state: generation 10838 (intent)

This commit is contained in:
2026-09-20 19:48:06 +00:00
parent 8a28e6a870
commit b00a3175c5
8 changed files with 1870 additions and 1653 deletions

View File

@@ -2146,7 +2146,7 @@
"taskType": "text-generation"
}
},
"generatedAt": "2026-09-20T19:33:34.912187+00:00",
"generatedAt": "2026-09-20T19:47:53.709356+00:00",
"summary": {
"activeBlockCount": 105,
"byGpuFramework": {

View File

@@ -425,121 +425,121 @@
}
},
"frameworkUpdatedAt": null,
"generatedAt": "2026-09-20T19:36:11.597557+00:00",
"generatedAt": "2026-09-20T19:47:53.721368+00:00",
"gpuStats": {
"Ascend_910-b3": {
"available": true,
"backlogHours": 924.0784313725491,
"backlogHours": 930.4342105263158,
"canVerify": true,
"error": null,
"gpu": "Ascend_910-b3",
"healthFactor": 1.0,
"maxConcurrentTasks": 8,
"qualityFactor": 0.8720057078999777,
"queueFactor": 0.999491884510734,
"queueWeight": 0.999491884510734,
"recentSuccess": 43,
"recentSuccessRate": 0.28104575163398693,
"recentTerminal": 153,
"recentWilsonLowerBound": 0.21585451117980445,
"qualityFactor": 0.7683607821382537,
"queueFactor": 0.9955475369627088,
"queueWeight": 0.9955475369627088,
"recentSuccess": 42,
"recentSuccessRate": 0.27631578947368424,
"recentTerminal": 152,
"recentWilsonLowerBound": 0.2114047281703244,
"running": 8,
"selectionWeight": 0.8715626282930653,
"selectionWeight": 0.764939684156479,
"stale": false,
"submissionEligible": true,
"throughputPerHour": 25.5,
"waiting": 23564
"throughputPerHour": 25.333333333333332,
"waiting": 23571
},
"Ascend_910-b4": {
"available": true,
"backlogHours": 1320.0,
"backlogHours": 1395.8360655737706,
"canVerify": true,
"error": null,
"gpu": "Ascend_910-b4",
"healthFactor": 1.0,
"maxConcurrentTasks": 8,
"qualityFactor": 1.8083922067820868,
"queueFactor": 0.9474351815026946,
"queueWeight": 0.9474351815026946,
"recentSuccess": 49,
"recentSuccessRate": 0.3798449612403101,
"recentTerminal": 129,
"recentWilsonLowerBound": 0.30071077692282255,
"qualityFactor": 1.7042352984745084,
"queueFactor": 0.9367844874016242,
"queueWeight": 0.9367844874016242,
"recentSuccess": 47,
"recentSuccessRate": 0.38524590163934425,
"recentTerminal": 122,
"recentWilsonLowerBound": 0.30364856037074855,
"running": 8,
"selectionWeight": 1.7133343986606449,
"selectionWeight": 1.5965011904931965,
"stale": false,
"submissionEligible": true,
"throughputPerHour": 21.5,
"waiting": 28380
"throughputPerHour": 20.333333333333332,
"waiting": 28382
},
"Biren_166m": {
"available": true,
"backlogHours": 197.96026490066225,
"backlogHours": 189.37974683544306,
"canVerify": true,
"error": null,
"gpu": "Biren_166m",
"healthFactor": 1.0,
"maxConcurrentTasks": 8,
"qualityFactor": 0.32165746706792336,
"queueFactor": 1.259357095038396,
"queueWeight": 1.259357095038396,
"recentSuccess": 29,
"recentSuccessRate": 0.19205298013245034,
"recentTerminal": 151,
"recentWilsonLowerBound": 0.1371784261383071,
"qualityFactor": 0.34688708845517613,
"queueFactor": 1.2640516591874138,
"queueWeight": 1.2640516591874138,
"recentSuccess": 32,
"recentSuccessRate": 0.20253164556962025,
"recentTerminal": 158,
"recentWilsonLowerBound": 0.1472736756110395,
"running": 8,
"selectionWeight": 0.4050816133240685,
"selectionWeight": 0.4384831997124566,
"stale": false,
"submissionEligible": true,
"throughputPerHour": 25.166666666666668,
"waiting": 4982
"throughputPerHour": 26.333333333333332,
"waiting": 4987
},
"Cambricon_mlu-370-x4": {
"available": true,
"backlogHours": 1768.9411764705883,
"backlogHours": 1769.0588235294117,
"canVerify": true,
"error": null,
"gpu": "Cambricon_mlu-370-x4",
"healthFactor": 1.0,
"maxConcurrentTasks": 7,
"qualityFactor": 1.2387892492962942,
"queueFactor": 0.9067312600185927,
"queueWeight": 0.9067312600185927,
"recentSuccess": 19,
"recentSuccessRate": 0.37254901960784315,
"qualityFactor": 1.3190863634264733,
"queueFactor": 0.9040730257623302,
"queueWeight": 0.9040730257623302,
"recentSuccess": 20,
"recentSuccessRate": 0.39215686274509803,
"recentTerminal": 51,
"recentWilsonLowerBound": 0.25320329720990103,
"recentWilsonLowerBound": 0.27027144858027696,
"running": 7,
"selectionWeight": 1.1232489369119154,
"selectionWeight": 1.1925503998248004,
"stale": false,
"submissionEligible": true,
"throughputPerHour": 8.5,
"waiting": 15036
"waiting": 15037
},
"Cambricon_mlu-370-x8": {
"available": true,
"backlogHours": 490.10526315789474,
"backlogHours": 454.4878048780488,
"canVerify": true,
"error": null,
"gpu": "Cambricon_mlu-370-x8",
"healthFactor": 1.0,
"maxConcurrentTasks": 7,
"qualityFactor": 0.2564701604785473,
"queueFactor": 1.0992391617214785,
"queueWeight": 1.0992391617214785,
"recentSuccess": 21,
"recentSuccessRate": 0.18421052631578946,
"recentTerminal": 114,
"recentWilsonLowerBound": 0.12375939628073249,
"qualityFactor": 0.2853901676156022,
"queueFactor": 1.108502093395098,
"queueWeight": 1.108502093395098,
"recentSuccess": 24,
"recentSuccessRate": 0.1951219512195122,
"recentTerminal": 123,
"recentWilsonLowerBound": 0.1347729686621224,
"running": 7,
"selectionWeight": 0.2819220442110114,
"selectionWeight": 0.31635559823627296,
"stale": false,
"submissionEligible": true,
"throughputPerHour": 19.0,
"waiting": 9312
"throughputPerHour": 20.5,
"waiting": 9317
},
"Iluvatar_bi-100": {
"available": true,
"backlogHours": 41.119113573407205,
"backlogHours": 40.81967213114754,
"canVerify": true,
"error": null,
"gpu": "Iluvatar_bi-100",
@@ -548,197 +548,197 @@
"qualityFactor": 0.05,
"queueFactor": 1.3,
"queueWeight": 1.3,
"recentSuccess": 23,
"recentSuccessRate": 0.06371191135734072,
"recentTerminal": 361,
"recentWilsonLowerBound": 0.04282606853752132,
"recentSuccess": 26,
"recentSuccessRate": 0.07103825136612021,
"recentTerminal": 366,
"recentWilsonLowerBound": 0.04893607879459068,
"running": 24,
"selectionWeight": 0.065,
"stale": false,
"submissionEligible": false,
"throughputPerHour": 60.166666666666664,
"waiting": 2474
"throughputPerHour": 61.0,
"waiting": 2490
},
"Iluvatar_bi-150": {
"available": true,
"backlogHours": 100.43389830508475,
"backlogHours": 102.26896551724137,
"canVerify": true,
"error": null,
"gpu": "Iluvatar_bi-150",
"healthFactor": 1.0,
"maxConcurrentTasks": 50,
"qualityFactor": 0.11679517071717828,
"qualityFactor": 0.0961242787103183,
"queueFactor": 1.3,
"queueWeight": 1.3,
"recentSuccess": 35,
"recentSuccessRate": 0.11864406779661017,
"recentTerminal": 295,
"recentWilsonLowerBound": 0.08655656627752173,
"running": 7,
"selectionWeight": 0.15183372193233177,
"recentSuccess": 33,
"recentSuccessRate": 0.11379310344827587,
"recentTerminal": 290,
"recentWilsonLowerBound": 0.0821829876510435,
"running": 6,
"selectionWeight": 0.1249615623234138,
"stale": false,
"submissionEligible": true,
"throughputPerHour": 49.166666666666664,
"waiting": 4938
"throughputPerHour": 48.333333333333336,
"waiting": 4943
},
"Iluvatar_mrv-100": {
"available": true,
"backlogHours": 3126.5806451612902,
"backlogHours": 3230.8,
"canVerify": true,
"error": null,
"gpu": "Iluvatar_mrv-100",
"healthFactor": 1.0,
"maxConcurrentTasks": 2,
"qualityFactor": 2.5,
"queueFactor": 0.8324825750946072,
"queueWeight": 0.8324825750946072,
"queueFactor": 0.8259777368641906,
"queueWeight": 0.8259777368641906,
"recentSuccess": 16,
"recentSuccessRate": 0.5161290322580645,
"recentTerminal": 31,
"recentWilsonLowerBound": 0.3484011828697317,
"recentSuccessRate": 0.5333333333333333,
"recentTerminal": 30,
"recentWilsonLowerBound": 0.36142013281525043,
"running": 2,
"selectionWeight": 2.081206437736518,
"selectionWeight": 2.0649443421604765,
"stale": false,
"submissionEligible": true,
"throughputPerHour": 5.166666666666667,
"throughputPerHour": 5.0,
"waiting": 16154
},
"Kunlunxin_p-800": {
"available": true,
"backlogHours": 585.1090909090909,
"backlogHours": 585.2181818181818,
"canVerify": true,
"error": null,
"gpu": "Kunlunxin_p-800",
"healthFactor": 1.0,
"maxConcurrentTasks": 8,
"qualityFactor": 0.3986199154808131,
"queueFactor": 1.0704097846258247,
"queueWeight": 1.0704097846258247,
"recentSuccess": 24,
"recentSuccessRate": 0.21818181818181817,
"qualityFactor": 0.4102582021056849,
"queueFactor": 1.0672525009913627,
"queueWeight": 1.0672525009913627,
"recentSuccess": 25,
"recentSuccessRate": 0.22727272727272727,
"recentTerminal": 110,
"recentWilsonLowerBound": 0.15122852088416214,
"recentWilsonLowerBound": 0.15894521607203976,
"running": 8,
"selectionWeight": 0.4266866578773816,
"selectionWeight": 0.4378490922495122,
"stale": false,
"submissionEligible": true,
"throughputPerHour": 18.333333333333332,
"waiting": 10727
"waiting": 10729
},
"MetaX_c-500": {
"available": true,
"backlogHours": 479.3093525179856,
"backlogHours": 469.30985915492954,
"canVerify": true,
"error": null,
"gpu": "MetaX_c-500",
"healthFactor": 1.0,
"maxConcurrentTasks": 4,
"qualityFactor": 0.20940249634895772,
"queueFactor": 1.1029179672601153,
"queueWeight": 1.1029179672601153,
"recentSuccess": 23,
"recentSuccessRate": 0.16546762589928057,
"recentTerminal": 139,
"recentWilsonLowerBound": 0.11286341507330147,
"qualityFactor": 0.20629180200750186,
"queueFactor": 1.1031787841741192,
"queueWeight": 1.1031787841741192,
"recentSuccess": 24,
"recentSuccessRate": 0.16901408450704225,
"recentTerminal": 142,
"recentWilsonLowerBound": 0.11628706631227818,
"running": 4,
"selectionWeight": 0.23095377561238614,
"selectionWeight": 0.22757673932372402,
"stale": false,
"submissionEligible": true,
"throughputPerHour": 23.166666666666668,
"waiting": 11104
"throughputPerHour": 23.666666666666668,
"waiting": 11107
},
"Mthreads_s4000": {
"available": true,
"backlogHours": 1778.4657534246576,
"backlogHours": 1829.1549295774646,
"canVerify": true,
"error": null,
"gpu": "Mthreads_s4000",
"healthFactor": 1.0,
"maxConcurrentTasks": 4,
"qualityFactor": 0.9095898145821663,
"queueFactor": 0.906001196453494,
"queueWeight": 0.906001196453494,
"qualityFactor": 0.8948711894200768,
"queueFactor": 0.8995540827536616,
"queueWeight": 0.8995540827536616,
"recentSuccess": 23,
"recentSuccessRate": 0.3150684931506849,
"recentTerminal": 73,
"recentWilsonLowerBound": 0.22003473675403795,
"recentSuccessRate": 0.323943661971831,
"recentTerminal": 71,
"recentWilsonLowerBound": 0.22657056384385282,
"running": 4,
"selectionWeight": 0.8240894602933544,
"selectionWeight": 0.8049850319814553,
"stale": false,
"submissionEligible": true,
"throughputPerHour": 12.166666666666666,
"waiting": 21638
"throughputPerHour": 11.833333333333334,
"waiting": 21645
},
"Sunrise_pt-200-x1": {
"available": true,
"backlogHours": 4270.909090909091,
"backlogHours": 4271.727272727273,
"canVerify": true,
"error": null,
"gpu": "Sunrise_pt-200-x1",
"healthFactor": 1.0,
"maxConcurrentTasks": 2,
"qualityFactor": 0.4739262052300172,
"queueFactor": 0.7944334976839765,
"queueWeight": 0.7944334976839765,
"recentSuccess": 7,
"recentSuccessRate": 0.3181818181818182,
"qualityFactor": 0.2703907318649671,
"queueFactor": 0.7920896256027817,
"queueWeight": 0.7920896256027817,
"recentSuccess": 6,
"recentSuccessRate": 0.2727272727272727,
"recentTerminal": 22,
"recentWilsonLowerBound": 0.16360387354692477,
"recentWilsonLowerBound": 0.13150582015814438,
"running": 2,
"selectionWeight": 0.37650285286497664,
"selectionWeight": 0.21417369356938393,
"stale": false,
"submissionEligible": true,
"throughputPerHour": 3.6666666666666665,
"waiting": 15660
"waiting": 15663
},
"Vastai_va16": {
"available": true,
"backlogHours": 1880.7391304347825,
"backlogHours": 1731.2399999999998,
"canVerify": true,
"error": null,
"gpu": "Vastai_va16",
"healthFactor": 1.0,
"maxConcurrentTasks": 16,
"qualityFactor": 0.05,
"queueFactor": 0.8984342778022876,
"queueWeight": 0.8984342778022876,
"queueFactor": 0.9070082996022887,
"queueWeight": 0.9070082996022887,
"recentSuccess": 4,
"recentSuccessRate": 0.08695652173913043,
"recentTerminal": 46,
"recentWilsonLowerBound": 0.03433531187670748,
"recentSuccessRate": 0.08,
"recentTerminal": 50,
"recentWilsonLowerBound": 0.031549003255790686,
"running": 16,
"selectionWeight": 0.04492171389011438,
"selectionWeight": 0.04535041498011444,
"stale": false,
"submissionEligible": false,
"throughputPerHour": 7.666666666666667,
"waiting": 14419
"throughputPerHour": 8.333333333333334,
"waiting": 14427
},
"hygon_k100-ai": {
"available": true,
"backlogHours": 917.8269230769231,
"backlogHours": 875.8899082568806,
"canVerify": true,
"error": null,
"gpu": "hygon_k100-ai",
"healthFactor": 1.0,
"maxConcurrentTasks": 6,
"qualityFactor": 0.620989901650416,
"queueFactor": 1.0005101026222207,
"queueWeight": 1.0005101026222207,
"recentSuccess": 27,
"recentSuccessRate": 0.25961538461538464,
"recentTerminal": 104,
"recentWilsonLowerBound": 0.1849887491976701,
"qualityFactor": 0.5669280073038301,
"queueFactor": 1.0046098323678758,
"queueWeight": 1.0046098323678758,
"recentSuccess": 28,
"recentSuccessRate": 0.25688073394495414,
"recentTerminal": 109,
"recentWilsonLowerBound": 0.18411863732206865,
"running": 6,
"selectionWeight": 0.6213066702276204,
"selectionWeight": 0.5695414503821546,
"stale": false,
"submissionEligible": true,
"throughputPerHour": 17.333333333333332,
"waiting": 15909
"throughputPerHour": 18.166666666666668,
"waiting": 15912
}
},
"queueAttemptedAt": "2026-09-20T19:33:34.924035+00:00",
"queueAttemptedAt": "2026-09-20T19:47:53.721368+00:00",
"queueError": null,
"queueUpdatedAt": "2026-09-20T19:33:34.924035+00:00",
"queueUpdatedAt": "2026-09-20T19:47:53.721368+00:00",
"supportedGpus": [
"Vastai_va16",
"Kunlunxin_p-800",

File diff suppressed because it is too large Load Diff

View File

@@ -1,6 +1,6 @@
{
"generatedAt": "2026-09-20T19:33:34.854966+00:00",
"lastSyncTime": "2026-09-20T19:33:34.552450+00:00",
"generatedAt": "2026-09-20T19:47:53.646354+00:00",
"lastSyncTime": "2026-09-20T19:47:52.978269+00:00",
"recentLimit": 300,
"report": {
"architectureCompatibilityBlocks": {
@@ -2567,28 +2567,28 @@
"unresolvedFailureCount": 0
},
"Biren_166m|vllm|text-generation": {
"attributableFailureCount": 37,
"decisionFailureRate": 0.7115,
"decisionSuccessRate": 0.2885,
"decisionTotal": 52,
"attributableFailureCount": 38,
"decisionFailureRate": 0.717,
"decisionSuccessRate": 0.283,
"decisionTotal": 53,
"failureBreakdown": {
"ambiguous_runtime": 3,
"ambiguous_runtime": 4,
"framework_architecture_unsupported": 28,
"model_load": 8,
"model_load": 9,
"tokenizer_compatibility": 1
},
"failureCount": 40,
"failureRate": 0.7273,
"failureCount": 42,
"failureRate": 0.7368,
"framework": "vllm",
"pendingCount": 0,
"pendingRate": 0.0,
"platformFailureCount": 0,
"successCount": 15,
"successRate": 0.2727,
"successRate": 0.2632,
"targetGpu": "Biren_166m",
"taskType": "text-generation",
"total": 55,
"unresolvedFailureCount": 3
"total": 57,
"unresolvedFailureCount": 4
},
"Cambricon_mlu-370-x4|unknown|text-generation": {
"attributableFailureCount": 695,
@@ -4563,7 +4563,7 @@
"decisionSuccessRate": 0.0,
"decisionTotal": 650,
"failureBreakdown": {
"ambiguous_runtime": 153,
"ambiguous_runtime": 154,
"architecture_compatibility": 30,
"attention_backend": 1,
"context_length": 42,
@@ -4575,7 +4575,7 @@
"runtime_memory": 53,
"tokenizer_compatibility": 79
},
"failureCount": 806,
"failureCount": 807,
"failureRate": 1.0,
"framework": "vllm",
"pendingCount": 0,
@@ -4585,8 +4585,8 @@
"successRate": 0.0,
"targetGpu": "hygon_k100-ai",
"taskType": "text-generation",
"total": 806,
"unresolvedFailureCount": 153
"total": 807,
"unresolvedFailureCount": 154
},
"hygon_k100-ai|vllm|unknown": {
"attributableFailureCount": 1,
@@ -4709,34 +4709,34 @@
"unresolvedFailureCount": 6406
},
"vllm": {
"attributableFailureCount": 3491,
"attributableFailureCount": 3492,
"decisionFailureRate": 0.9771,
"decisionSuccessRate": 0.0229,
"decisionTotal": 3573,
"decisionTotal": 3574,
"failureBreakdown": {
"ambiguous_runtime": 1440,
"ambiguous_runtime": 1442,
"architecture_compatibility": 112,
"attention_backend": 1,
"backend_operator": 84,
"context_length": 161,
"framework_architecture_unsupported": 1272,
"memory_capacity": 720,
"model_load": 200,
"model_load": 201,
"platform_infrastructure": 864,
"repository_structure": 470,
"runtime_memory": 62,
"tokenizer_compatibility": 409,
"参数/模板问题": 40
},
"failureCount": 5835,
"failureCount": 5838,
"failureRate": 0.9861,
"pendingCount": 0,
"pendingRate": 0.0,
"platformFailureCount": 864,
"successCount": 82,
"successRate": 0.0139,
"total": 5917,
"unresolvedFailureCount": 1480
"total": 5920,
"unresolvedFailureCount": 1482
},
"vllm-customized": {
"attributableFailureCount": 1,
@@ -4870,7 +4870,7 @@
"unresolvedFailureCount": 32
}
},
"generatedAt": "2026-09-20T19:33:34.846545+00:00",
"generatedAt": "2026-09-20T19:47:53.636798+00:00",
"gpuSummaries": {
"Ascend_910-b3": {
"attributableFailureCount": 78,
@@ -4926,17 +4926,17 @@
"unresolvedFailureCount": 591
},
"Biren_166m": {
"attributableFailureCount": 175,
"decisionFailureRate": 0.875,
"decisionSuccessRate": 0.125,
"decisionTotal": 200,
"attributableFailureCount": 176,
"decisionFailureRate": 0.8756,
"decisionSuccessRate": 0.1244,
"decisionTotal": 201,
"failureBreakdown": {
"ambiguous_runtime": 133,
"ambiguous_runtime": 134,
"backend_operator": 4,
"context_length": 10,
"framework_architecture_unsupported": 115,
"memory_capacity": 1,
"model_load": 19,
"model_load": 20,
"platform_infrastructure": 2,
"repository_structure": 18,
"runtime_memory": 2,
@@ -4945,15 +4945,15 @@
"日志缺失": 62,
"验证失败": 27
},
"failureCount": 613,
"failureRate": 0.9608,
"failureCount": 615,
"failureRate": 0.9609,
"pendingCount": 0,
"pendingRate": 0.0,
"platformFailureCount": 2,
"successCount": 25,
"successRate": 0.0392,
"total": 638,
"unresolvedFailureCount": 436
"successRate": 0.0391,
"total": 640,
"unresolvedFailureCount": 437
},
"Cambricon_mlu-370-x4": {
"attributableFailureCount": 711,
@@ -5257,7 +5257,7 @@
"decisionSuccessRate": 0.0417,
"decisionTotal": 719,
"failureBreakdown": {
"ambiguous_runtime": 298,
"ambiguous_runtime": 299,
"architecture_compatibility": 32,
"attention_backend": 1,
"context_length": 46,
@@ -5272,15 +5272,15 @@
"日志缺失": 66,
"验证失败": 27
},
"failureCount": 1460,
"failureCount": 1461,
"failureRate": 0.9799,
"pendingCount": 0,
"pendingRate": 0.0,
"platformFailureCount": 4,
"successCount": 30,
"successRate": 0.0201,
"total": 1490,
"unresolvedFailureCount": 767
"total": 1491,
"unresolvedFailureCount": 768
}
},
"observedGpuMemoryGiB": {
@@ -6244,6 +6244,29 @@
"total": 2,
"unresolvedFailureCount": 1
},
"Biren_166m|vllm|text-generation|mpt|none": {
"attributableFailureCount": 1,
"decisionFailureRate": 1.0,
"decisionSuccessRate": 0.0,
"decisionTotal": 1,
"failureBreakdown": {
"model_load": 1
},
"failureCount": 1,
"failureRate": 1.0,
"framework": "vllm",
"modelType": "mpt",
"pendingCount": 0,
"pendingRate": 0.0,
"platformFailureCount": 0,
"quantizationMethod": "none",
"successCount": 0,
"successRate": 0.0,
"targetGpu": "Biren_166m",
"taskType": "text-generation",
"total": 1,
"unresolvedFailureCount": 0
},
"Biren_166m|vllm|text-generation|muse_glimmer|none": {
"attributableFailureCount": 1,
"decisionFailureRate": 1.0,
@@ -6336,6 +6359,29 @@
"total": 1,
"unresolvedFailureCount": 0
},
"Biren_166m|vllm|text-generation|qwen2|compressed-tensors": {
"attributableFailureCount": 0,
"decisionFailureRate": 0.0,
"decisionSuccessRate": 0.0,
"decisionTotal": 0,
"failureBreakdown": {
"ambiguous_runtime": 1
},
"failureCount": 1,
"failureRate": 1.0,
"framework": "vllm",
"modelType": "qwen2",
"pendingCount": 0,
"pendingRate": 0.0,
"platformFailureCount": 0,
"quantizationMethod": "compressed-tensors",
"successCount": 0,
"successRate": 0.0,
"targetGpu": "Biren_166m",
"taskType": "text-generation",
"total": 1,
"unresolvedFailureCount": 1
},
"Biren_166m|vllm|text-generation|qwen2|none": {
"attributableFailureCount": 0,
"decisionFailureRate": 0.0,
@@ -12195,6 +12241,29 @@
"taskType": "text-generation",
"total": 1,
"unresolvedFailureCount": 0
},
"hygon_k100-ai|vllm|text-generation|qwen3|none": {
"attributableFailureCount": 0,
"decisionFailureRate": 0.0,
"decisionSuccessRate": 0.0,
"decisionTotal": 0,
"failureBreakdown": {
"ambiguous_runtime": 1
},
"failureCount": 1,
"failureRate": 1.0,
"framework": "vllm",
"modelType": "qwen3",
"pendingCount": 0,
"pendingRate": 0.0,
"platformFailureCount": 0,
"quantizationMethod": "none",
"successCount": 0,
"successRate": 0.0,
"targetGpu": "hygon_k100-ai",
"taskType": "text-generation",
"total": 1,
"unresolvedFailureCount": 1
}
},
"recentCombinationStats": {
@@ -12382,31 +12451,31 @@
"unresolvedFailureCount": 1
},
"Biren_166m|vllm|text-generation": {
"attributableFailureCount": 4,
"consecutiveFailures": 3,
"attributableFailureCount": 5,
"consecutiveFailures": 4,
"consecutivePlatformFailures": 0,
"decisionFailureRate": 0.8,
"decisionSuccessRate": 0.2,
"decisionTotal": 5,
"decisionFailureRate": 0.8333,
"decisionSuccessRate": 0.1667,
"decisionTotal": 6,
"failureBreakdown": {
"ambiguous_runtime": 2,
"ambiguous_runtime": 3,
"framework_architecture_unsupported": 3,
"model_load": 1
"model_load": 2
},
"failureCount": 6,
"failureRate": 0.8571,
"failureCount": 8,
"failureRate": 0.8889,
"framework": "vllm",
"lastPlatformFailureAt": null,
"lastTerminalAt": "2026-09-20T15:32:19.355892+00:00",
"lastTerminalAt": "2026-09-20T19:47:52.978269+00:00",
"pendingCount": 0,
"pendingRate": 0.0,
"platformFailureCount": 0,
"successCount": 1,
"successRate": 0.1429,
"successRate": 0.1111,
"targetGpu": "Biren_166m",
"taskType": "text-generation",
"total": 7,
"unresolvedFailureCount": 2
"total": 9,
"unresolvedFailureCount": 3
},
"Cambricon_mlu-370-x4|unknown|text-generation": {
"attributableFailureCount": 7,
@@ -12654,9 +12723,9 @@
"decisionSuccessRate": 0.0,
"decisionTotal": 0,
"failureBreakdown": {
"ambiguous_runtime": 8
"ambiguous_runtime": 7
},
"failureCount": 8,
"failureCount": 7,
"failureRate": 1.0,
"framework": "unknown",
"lastPlatformFailureAt": null,
@@ -12668,8 +12737,8 @@
"successRate": 0.0,
"targetGpu": "Kunlunxin_p-800",
"taskType": "text-generation",
"total": 8,
"unresolvedFailureCount": 8
"total": 7,
"unresolvedFailureCount": 7
},
"Kunlunxin_p-800|vllm_fix_tokenizer|text-generation": {
"attributableFailureCount": 0,
@@ -13044,11 +13113,11 @@
"decisionSuccessRate": 0.0,
"decisionTotal": 10,
"failureBreakdown": {
"ambiguous_runtime": 6,
"ambiguous_runtime": 5,
"framework_architecture_unsupported": 9,
"runtime_memory": 1
},
"failureCount": 16,
"failureCount": 15,
"failureRate": 1.0,
"framework": "vllm",
"lastPlatformFailureAt": null,
@@ -13060,8 +13129,8 @@
"successRate": 0.0,
"targetGpu": "hygon_k100-ai",
"taskType": "text-generation",
"total": 16,
"unresolvedFailureCount": 6
"total": 15,
"unresolvedFailureCount": 5
}
},
"recentProfileCombinationStats": {
@@ -13391,6 +13460,31 @@
"total": 2,
"unresolvedFailureCount": 1
},
"Biren_166m|vllm|text-generation|mpt|none": {
"attributableFailureCount": 1,
"consecutiveFailures": 1,
"decisionFailureRate": 1.0,
"decisionSuccessRate": 0.0,
"decisionTotal": 1,
"failureBreakdown": {
"model_load": 1
},
"failureCount": 1,
"failureRate": 1.0,
"framework": "vllm",
"lastTerminalAt": "2026-09-20T19:47:52.978254+00:00",
"modelType": "mpt",
"pendingCount": 0,
"pendingRate": 0.0,
"platformFailureCount": 0,
"quantizationMethod": "none",
"successCount": 0,
"successRate": 0.0,
"targetGpu": "Biren_166m",
"taskType": "text-generation",
"total": 1,
"unresolvedFailureCount": 0
},
"Biren_166m|vllm|text-generation|nemotron_h|modelopt": {
"attributableFailureCount": 1,
"consecutiveFailures": 1,
@@ -13416,6 +13510,31 @@
"total": 1,
"unresolvedFailureCount": 0
},
"Biren_166m|vllm|text-generation|qwen2|compressed-tensors": {
"attributableFailureCount": 0,
"consecutiveFailures": 0,
"decisionFailureRate": 0.0,
"decisionSuccessRate": 0.0,
"decisionTotal": 0,
"failureBreakdown": {
"ambiguous_runtime": 1
},
"failureCount": 1,
"failureRate": 1.0,
"framework": "vllm",
"lastTerminalAt": "2026-09-20T19:47:52.978269+00:00",
"modelType": "qwen2",
"pendingCount": 0,
"pendingRate": 0.0,
"platformFailureCount": 0,
"quantizationMethod": "compressed-tensors",
"successCount": 0,
"successRate": 0.0,
"targetGpu": "Biren_166m",
"taskType": "text-generation",
"total": 1,
"unresolvedFailureCount": 1
},
"Biren_166m|vllm|text-generation|qwen3|none": {
"attributableFailureCount": 0,
"consecutiveFailures": 0,
@@ -15367,6 +15486,31 @@
"taskType": "text-generation",
"total": 1,
"unresolvedFailureCount": 0
},
"hygon_k100-ai|vllm|text-generation|qwen3|none": {
"attributableFailureCount": 0,
"consecutiveFailures": 0,
"decisionFailureRate": 0.0,
"decisionSuccessRate": 0.0,
"decisionTotal": 0,
"failureBreakdown": {
"ambiguous_runtime": 1
},
"failureCount": 1,
"failureRate": 1.0,
"framework": "vllm",
"lastTerminalAt": "2026-09-20T19:47:52.978239+00:00",
"modelType": "qwen3",
"pendingCount": 0,
"pendingRate": 0.0,
"platformFailureCount": 0,
"quantizationMethod": "none",
"successCount": 0,
"successRate": 0.0,
"targetGpu": "hygon_k100-ai",
"taskType": "text-generation",
"total": 1,
"unresolvedFailureCount": 1
}
},
"sizedProfileCombinationStats": {
@@ -16753,6 +16897,30 @@
"total": 2,
"unresolvedFailureCount": 1
},
"Biren_166m|vllm|text-generation|mpt|none|32": {
"attributableFailureCount": 1,
"decisionFailureRate": 1.0,
"decisionSuccessRate": 0.0,
"decisionTotal": 1,
"failureBreakdown": {
"model_load": 1
},
"failureCount": 1,
"failureRate": 1.0,
"framework": "vllm",
"loadSizeLog2Bucket": 32,
"modelType": "mpt",
"pendingCount": 0,
"pendingRate": 0.0,
"platformFailureCount": 0,
"quantizationMethod": "none",
"successCount": 0,
"successRate": 0.0,
"targetGpu": "Biren_166m",
"taskType": "text-generation",
"total": 1,
"unresolvedFailureCount": 0
},
"Biren_166m|vllm|text-generation|muse_glimmer|none|34": {
"attributableFailureCount": 1,
"decisionFailureRate": 1.0,
@@ -16849,6 +17017,30 @@
"total": 1,
"unresolvedFailureCount": 0
},
"Biren_166m|vllm|text-generation|qwen2|compressed-tensors|29": {
"attributableFailureCount": 0,
"decisionFailureRate": 0.0,
"decisionSuccessRate": 0.0,
"decisionTotal": 0,
"failureBreakdown": {
"ambiguous_runtime": 1
},
"failureCount": 1,
"failureRate": 1.0,
"framework": "vllm",
"loadSizeLog2Bucket": 29,
"modelType": "qwen2",
"pendingCount": 0,
"pendingRate": 0.0,
"platformFailureCount": 0,
"quantizationMethod": "compressed-tensors",
"successCount": 0,
"successRate": 0.0,
"targetGpu": "Biren_166m",
"taskType": "text-generation",
"total": 1,
"unresolvedFailureCount": 1
},
"Biren_166m|vllm|text-generation|qwen2|none|31": {
"attributableFailureCount": 0,
"decisionFailureRate": 0.0,
@@ -25581,24 +25773,48 @@
"taskType": "text-generation",
"total": 1,
"unresolvedFailureCount": 0
},
"hygon_k100-ai|vllm|text-generation|qwen3|none|33": {
"attributableFailureCount": 0,
"decisionFailureRate": 0.0,
"decisionSuccessRate": 0.0,
"decisionTotal": 0,
"failureBreakdown": {
"ambiguous_runtime": 1
},
"failureCount": 1,
"failureRate": 1.0,
"framework": "vllm",
"loadSizeLog2Bucket": 33,
"modelType": "qwen3",
"pendingCount": 0,
"pendingRate": 0.0,
"platformFailureCount": 0,
"quantizationMethod": "none",
"successCount": 0,
"successRate": 0.0,
"targetGpu": "hygon_k100-ai",
"taskType": "text-generation",
"total": 1,
"unresolvedFailureCount": 1
}
},
"terminalRecords": 15939,
"totalRecords": 16073,
"terminalRecords": 15942,
"totalRecords": 16076,
"totals": {
"attributableFailureCount": 5788,
"decisionFailureRate": 0.8614,
"decisionSuccessRate": 0.1386,
"decisionTotal": 6719,
"attributableFailureCount": 5789,
"decisionFailureRate": 0.8615,
"decisionSuccessRate": 0.1385,
"decisionTotal": 6720,
"failureBreakdown": {
"ambiguous_runtime": 3796,
"ambiguous_runtime": 3798,
"architecture_compatibility": 212,
"attention_backend": 1,
"backend_operator": 102,
"context_length": 318,
"framework_architecture_unsupported": 2006,
"memory_capacity": 1196,
"model_load": 483,
"model_load": 484,
"platform_infrastructure": 922,
"repository_structure": 734,
"runtime_memory": 81,
@@ -25607,15 +25823,15 @@
"日志缺失": 719,
"验证失败": 673
},
"failureCount": 15008,
"failureCount": 15011,
"failureRate": 0.9416,
"pendingCount": 0,
"pendingRate": 0.0,
"platformFailureCount": 922,
"successCount": 931,
"successRate": 0.0584,
"total": 15939,
"unresolvedFailureCount": 8298
"total": 15942,
"unresolvedFailureCount": 8300
},
"warnings": [
"GPU Iluvatar_bi-100 本地统计失败率偏高≥50%),建议重点关注。",
@@ -25660,12 +25876,12 @@
"组合 Cambricon_mlu-370-x4|unknown|visual-multi-modal 近期失败集中,建议降低该 GPU+框架的提交优先级。",
"组合 Ascend_910-b4|vllm|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。",
"组合 Iluvatar_bi-150|transformers|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。",
"组合 Cambricon_mlu-370-x8|vllm|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。",
"组合 Cambricon_mlu-370-x8|unknown|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。",
"组合 Ascend_910-b3|vllm|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。",
"组合 Cambricon_mlu-370-x8|vllm|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。"
"组合 Ascend_910-b3|vllm|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。"
]
},
"storageMode": "decision_state_only",
"summarizedRecords": 16073,
"summarizedRecords": 16076,
"version": 1
}

View File

@@ -35,6 +35,7 @@
{"failReason": "tokenizer_compatibility", "failureAction": "check_tokenizer_files_then_llm", "failureCategory": "tokenizer_compatibility", "failureClassificationReason": "broad_tokenizer_error", "failureCode": "TOKENIZER_FAILED", "failureDetectedFramework": "transformers", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "model_framework", "framework": "transformers", "lastSyncTime": "2026-09-20T15:35:51.777956+00:00", "modelId": "LiquidAI/LFM2.5-8B-A1B", "modelProfile": {"architectures": ["Lfm2MoeForCausalLM"], "configFingerprint": "3086a71adcdca90005b4967eb30749c44cd82389aff8d599758e4b91675d40dc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 16936006912, "estimatedRequiredGiB": 18.948, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2_moe", "modelscopeFileSize": 16953948786, "modelscopeLicense": "other", "modelscopeParams": 8467856832, "modelscopeTags": ["license:other", "model_type:lfm2_moe", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 16953948786}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T07:24:14.091402+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4973030", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["qwen3_5"], "framework": "", "lastSyncTime": "2026-09-20T06:27:54.872472+00:00", "modelId": "logic65/Qwen3.8-Whittle-w50-18.3B-unrepaired", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T06:21:21+00:00", "targetGpu": "Biren_166m", "taskId": "4610365", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDetectedFramework": "vllm_fix_tokenizer", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-20T14:47:45.260018+00:00", "modelId": "inclusionAI/Ling-3.0-tiny", "modelProfile": {"architectures": ["BailingMoeV3ForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 15787992416, "estimatedRequiredGiB": 17.659, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "bailing_hybrid", "modelscopeFileSize": 15801179822, "modelscopeLicense": "mit", "modelscopeParams": 7893392800, "modelscopeTags": ["license:mit", "model_type:bailing_hybrid", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 15801179822}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T06:19:35.394755+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4972231", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-20T19:47:52.978269+00:00", "modelId": "neuralmagic/Qwen2-0.5B-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["Qwen2ForCausalLM"], "configFingerprint": "d341b8002cf39d3299fd429479e8c8e954583e49d4fec5a640ddc0f8f0cbd054", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 903168128, "estimatedRequiredGiB": 1.022, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen2", "modelscopeFileSize": 914658310, "modelscopeLicense": "apache-2.0", "modelscopeParams": 630167424, "modelscopeTags": ["license:apache-2.0", "model_type:qwen2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 914658310}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T06:17:57.834294+00:00", "targetGpu": "Biren_166m", "taskId": "4972230", "taskType": "text-generation", "verifyResult": -1}
{"failReason": null, "framework": "", "lastSyncTime": "2026-09-20T05:59:18.862303+00:00", "modelId": "AI-ModelScope/granite-20b-code-base-8k", "modelProfile": {}, "outcome": "success", "status": "success", "submitTime": "2026-09-20T05:57:22+00:00", "targetGpu": "MetaX_c-500", "taskId": "4079080", "taskType": "text-generation", "verifyResult": 1}
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDetectedFramework": "transformers", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "framework": "transformers", "lastSyncTime": "2026-09-20T13:30:42.155956+00:00", "modelId": "Youssofal/Qwen3.5-9B-MTPLX-Optimized-Speed", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "3086a71adcdca90005b4967eb30749c44cd82389aff8d599758e4b91675d40dc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 8674988799, "estimatedRequiredGiB": 9.718, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 8695119291, "modelscopeLicense": "apache-2.0", "modelscopeParams": 2415484144, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:apple-silicon", "custom_tag:speculative-decoding", "custom_tag:qwen", "custom_tag:qwen3", "custom_tag:mtp", "custom_tag:mtplx", "custom_tag:local-ai", "custom_tag:qwen3-5", "custom_tag:9b", "custom_tag:6-bit"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 8695119291}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T05:16:13.358703+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4971314", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDetectedFramework": "vllm", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "framework": "vllm", "lastSyncTime": "2026-09-20T13:30:42.156020+00:00", "modelId": "BAAI/RoboBrain2.5-8B-MT", "modelProfile": {"architectures": ["Qwen3VLForConditionalGeneration"], "configFingerprint": "ac4b2845ddd4bdce478dd2b65134ee3ab3d259a119afe6841bf068b135fd4892", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 17534339552, "estimatedRequiredGiB": 19.609, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_vl", "modelscopeFileSize": 17545918174, "modelscopeLicense": "apache-2.0", "modelscopeParams": 8767123696, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_vl", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 17545918174}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T05:16:13.351034+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4971313", "taskType": "text-generation", "verifyResult": -1}
@@ -66,6 +67,7 @@
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDetectedFramework": "transformers", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "transformers", "lastSyncTime": "2026-09-20T12:34:29.856361+00:00", "modelId": "LiquidAI/LFM2.5-1.2B-Thinking-MLX-6bit", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "3086a71adcdca90005b4967eb30749c44cd82389aff8d599758e4b91675d40dc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 951093016, "estimatedRequiredGiB": 1.068, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 955963132, "modelscopeLicense": "other", "modelscopeParams": 256113408, "modelscopeTags": ["license:other", "model_type:lfm2", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge", "custom_tag:mlx", "custom_tag:reasoning"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 955963132}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T04:14:41.846486+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4970590", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_0_17_0_corex_4_4_0", "lastSyncTime": "2026-09-20T12:58:28.353096+00:00", "modelId": "RedHatAI/starcoder2-15b-quantized.w8a16", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "1cc16a7df03cf7b1485e2fda7e5c05e286a546d04e80bdd595c015004b3852a8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 17785269848, "estimatedRequiredGiB": 19.88, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 17788663591, "modelscopeLicense": "bigcode-openrail-m", "modelscopeParams": 15957889024, "modelscopeTags": ["license:bigcode-openrail-m", "model_type:starcoder2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 17788663591}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T04:09:56.550329+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4970503", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "model_load", "failureAction": "inspect_root_exception_then_llm_if_unknown", "failureCategory": "model_load", "failureClassificationReason": "broad_load_error", "failureCode": "MODEL_LOAD_FAILED", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "model_gpu_framework", "framework": "vllm", "lastSyncTime": "2026-09-20T15:32:19.355892+00:00", "modelId": "neuralmagic/Mistral-Nemo-Instruct-2407-quantized.w4a16", "modelProfile": {"architectures": ["MistralForCausalLM"], "configFingerprint": "d341b8002cf39d3299fd429479e8c8e954583e49d4fec5a640ddc0f8f0cbd054", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 8308282608, "estimatedRequiredGiB": 9.319, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "mistral", "modelscopeFileSize": 8338221588, "modelscopeLicense": "llama2", "modelscopeParams": 12247782400, "modelscopeTags": ["license:llama2", "model_type:mistral", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 8338221588}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T04:09:56.548188+00:00", "targetGpu": "Biren_166m", "taskId": "4970505", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "model_load", "failureAction": "inspect_root_exception_then_llm_if_unknown", "failureCategory": "model_load", "failureClassificationReason": "broad_load_error", "failureCode": "MODEL_LOAD_FAILED", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "model_gpu_framework", "framework": "vllm", "lastSyncTime": "2026-09-20T19:47:52.978254+00:00", "modelId": "aisingapore/SEA-LION-v1-7B-IT-GPTQ", "modelProfile": {"architectures": ["MPTForCausalLM"], "configFingerprint": "f7d2aee0bcf1396cd8ab3b1064b6a032eccc04f2d655a336023e13528574903d", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 5469228368, "estimatedRequiredGiB": 6.118, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "mpt", "modelscopeFileSize": 5473985594, "modelscopeLicense": "mit", "modelscopeParams": 1922048000, "modelscopeTags": ["license:mit", "model_type:mpt", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 5473985594}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T04:09:56.541899+00:00", "targetGpu": "Biren_166m", "taskId": "4970509", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_0_17_0_corex_4_4_0", "lastSyncTime": "2026-09-20T12:34:29.856396+00:00", "modelId": "aisingapore/SEA-LION-v1-7B-IT-Research", "modelProfile": {"architectures": ["MPTForCausalLM"], "configFingerprint": "17a693aa8ae65d96fb0c459d2e1f98ce1d8b2f8933e06b75903df1a5b48287eb", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 15003361968, "estimatedRequiredGiB": 16.773, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "mpt", "modelscopeFileSize": 15008118474, "modelscopeLicense": "cc-by-nc-sa-4.0", "modelscopeParams": 7501651968, "modelscopeTags": ["license:cc-by-nc-sa-4.0", "model_type:mpt", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 15008118474}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T04:09:56.540284+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4970507", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_0_17_0_corex_4_4_0", "lastSyncTime": "2026-09-20T12:34:29.856340+00:00", "modelId": "Arain119/Sophia", "modelProfile": {"architectures": ["SophiaForCausalLM"], "configFingerprint": "1cc16a7df03cf7b1485e2fda7e5c05e286a546d04e80bdd595c015004b3852a8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2226652224, "estimatedRequiredGiB": 7.28, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "sophia_hybrid", "modelscopeFileSize": 6513869659, "modelscopeLicense": "Apache License 2.0", "modelscopeParams": 1113293776, "modelscopeTags": ["license:Apache License 2.0", "model_type:sophia_hybrid", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 6513869659}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T04:09:56.493446+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4970510", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["zaya"], "framework": "vllm", "lastSyncTime": "2026-09-20T17:09:45.266103+00:00", "modelId": "JANGQ-AI/Osaurus-AppleScript-8B-JANG_6M", "modelProfile": {"architectures": ["ZayaForCausalLM"], "configFingerprint": "d341b8002cf39d3299fd429479e8c8e954583e49d4fec5a640ddc0f8f0cbd054", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7915599888, "estimatedRequiredGiB": 8.885, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "zaya", "modelscopeFileSize": 7950463197, "modelscopeLicense": "other", "modelscopeParams": 2274126328, "modelscopeTags": ["license:other", "model_type:zaya", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:applescript", "custom_tag:macos", "custom_tag:computer-use", "custom_tag:agent", "custom_tag:tool-calling", "custom_tag:function-calling", "custom_tag:mlx", "custom_tag:moe", "custom_tag:osaurus"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 7950463197}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T04:02:45.346171+00:00", "targetGpu": "Biren_166m", "taskId": "4970445", "taskType": "text-generation", "verifyResult": -1}
@@ -124,6 +126,7 @@
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T14:32:15.260751+00:00", "modelId": "dickeyli/llmrec-onereason-8b-sh2-grpo4-step160", "modelProfile": {"architectures": ["Qwen3ForCausalLM"], "configFingerprint": "d341b8002cf39d3299fd429479e8c8e954583e49d4fec5a640ddc0f8f0cbd054", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 16779959304, "estimatedRequiredGiB": 18.777, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3", "modelscopeFileSize": 16801630203, "modelscopeLicense": "Apache License 2.0", "modelscopeParams": 8389956608, "modelscopeTags": ["license:Apache License 2.0", "model_type:qwen3", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 16801630203}, "outcome": "success", "status": "success", "submitTime": "2026-09-20T03:09:18.440547+00:00", "targetGpu": "Biren_166m", "taskId": "4969359", "taskType": "text-generation", "verifyResult": 1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "driver_error", "failureCode": "DRIVER_ERROR", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T18:18:38.666111+00:00", "modelId": "neuralmagic/SmolLM-135M-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 219850592, "estimatedRequiredGiB": 0.249, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 223236742, "modelscopeLicense": "apache-2.0", "modelscopeParams": 162826560, "modelscopeTags": ["license:apache-2.0", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 223236742}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T03:09:11.539809+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4969300", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "driver_error", "failureCode": "DRIVER_ERROR", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T18:18:38.666090+00:00", "modelId": "LiquidAI/LFM2.5-2.6B", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 5394427456, "estimatedRequiredGiB": 6.049, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 5412393192, "modelscopeLicense": "other", "modelscopeParams": 2697198592, "modelscopeTags": ["license:other", "model_type:lfm2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 5412393192}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T03:09:11.537727+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4969303", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "missing_structured_error_code", "failureCode": null, "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-20T19:47:52.978239+00:00", "modelId": "dickeyli/llmrec-onereason-8b-sh2-grpo4-step160", "modelProfile": {"architectures": ["Qwen3ForCausalLM"], "configFingerprint": "8b3c0a82b858cf26b1a0f6075074234a66487d5734741dbbfd4743963b967faa", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 16779959304, "estimatedRequiredGiB": 18.777, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3", "modelscopeFileSize": 16801630203, "modelscopeLicense": "Apache License 2.0", "modelscopeParams": 8389956608, "modelscopeTags": ["license:Apache License 2.0", "model_type:qwen3", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 16801630203}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T02:52:52.295134+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4969090", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "model_load", "failureAction": "inspect_root_exception_then_llm_if_unknown", "failureCategory": "model_load", "failureClassificationReason": "broad_load_error", "failureCode": "MODEL_LOAD_FAILED", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "model_gpu_framework", "framework": "vllm-patch-tokenizer", "lastSyncTime": "2026-09-20T19:19:40.458840+00:00", "modelId": "LiquidAI/LFM2.5-1.2B-Thinking-MLX-5bit", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "0d5bc82787a89d4313672808d1a8629a1fc675c91aa6a08865f770aa24447fbc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 804816628, "estimatedRequiredGiB": 0.905, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 809686744, "modelscopeLicense": "other", "modelscopeParams": 219544320, "modelscopeTags": ["license:other", "model_type:lfm2", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge", "custom_tag:mlx", "custom_tag:reasoning"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 809686744}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T02:52:52.134762+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4969081", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDetectedFramework": "vllm", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["qwen3_5_moe"], "framework": "vllm-patch-tokenizer", "lastSyncTime": "2026-09-20T18:53:01.473075+00:00", "modelId": "mlx-community/XYZ-Aquila-mini-OptiQ-4bit", "modelProfile": {"architectures": ["Qwen3_5MoeForConditionalGeneration"], "configFingerprint": "0d5bc82787a89d4313672808d1a8629a1fc675c91aa6a08865f770aa24447fbc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 37184031394, "estimatedRequiredGiB": 41.579, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_moe", "modelscopeFileSize": 37204398600, "modelscopeLicense": "apache-2.0", "modelscopeParams": 10061358960, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5_moe", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:optiq", "custom_tag:mixed-precision", "custom_tag:qwen3.6", "custom_tag:agentic-search"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 37204398600}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T02:52:52.043508+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4969075", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedArchitectures": ["Cohere2ForCausalLM"], "framework": "vllm", "lastSyncTime": "2026-09-20T17:09:45.266177+00:00", "modelId": "CohereLabs/tiny-aya-en-thinker", "modelProfile": {"architectures": ["Cohere2ForCausalLM"], "configFingerprint": "6db01b5037d5ed2db74875d2113f8582c97f34bd825637161e54174c9f2cabed", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 6700539432, "estimatedRequiredGiB": 7.522, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "cohere2", "modelscopeFileSize": 6730845002, "modelscopeLicense": "cc-by-nc-4.0", "modelscopeParams": 3350243328, "modelscopeTags": ["license:cc-by-nc-4.0", "model_type:cohere2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:aya", "custom_tag:multilingual", "custom_tag:reasoning", "custom_tag:tiny-aya"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 6730845002}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T02:52:51.937617+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4969071", "taskType": "text-generation", "verifyResult": -1}
@@ -295,6 +298,3 @@
{"failReason": "tokenizer_compatibility", "failureAction": "check_tokenizer_files_then_llm", "failureCategory": "tokenizer_compatibility", "failureClassificationReason": "broad_tokenizer_error", "failureCode": "TOKENIZER_FAILED", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "model_framework", "framework": "", "lastSyncTime": "2026-09-15T01:15:42.313690+00:00", "modelId": "zstack/qwen3-4b-fake", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-15T01:15:22+00:00", "targetGpu": "Iluvatar_mrv-100", "taskId": "4588142", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["qwen3_5"], "framework": "", "lastSyncTime": "2026-09-15T00:56:25.725574+00:00", "modelId": "callmezcc/Qwen3.8-27B-GPTQ-W4A16", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-15T00:53:21+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "4332632", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDetectedFramework": "vllm", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["qwen3_5"], "framework": "vllm", "lastSyncTime": "2026-09-15T00:40:39.932460+00:00", "modelId": "cyankiwi/Ornith-1.5-9B-AWQ-FP8", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-15T00:33:21+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4591815", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "", "lastSyncTime": "2026-09-15T00:31:34.841419+00:00", "modelId": "nanbeige/Nanbeige4.2-3B-GPTQ-Int8", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-15T00:31:21+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4597481", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "unknown", "failureCode": "UNKNOWN", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-15T00:22:04.249561+00:00", "modelId": "FlagRelease/DeepSeek-R1-Distill-Qwen-1.5B-mthreads-FlagOS", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-15T00:19:21+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4592892", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "unknown", "failureCode": "UNKNOWN", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-15T00:22:04.249539+00:00", "modelId": "solidrust/Llama-3-8B-Instruct-v0.4-AWQ", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-15T00:17:21+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4595877", "taskType": "text-generation", "verifyResult": -1}

View File

@@ -1759,6 +1759,32 @@
{"batchId": "91968e61747d4364bdf7fbbd10ba49db", "completedAt": "2026-09-20T19:33:29.660592+00:00", "configFingerprint": "2e9cebc01b20081f0839fa18fe9ca4f7d059509f235699027201fe328177b304", "configSource": "modelhub_live", "createdAt": "2026-09-20T19:30:13.722451+00:00", "framework": "vllm_fix_tokenizer", "intentId": "b323cdb4c8604c8f827ffd1eb1b9811b", "lastModified": "2026-08-31T14:27:40+00:00", "modelAddress": "https://modelscope.cn/models/sbintuitions/sarashina2.2-3b-instruct-v0.1", "reason": null, "reconciledAt": "2026-09-20T19:47:50.163977+00:00", "repoId": "sbintuitions/sarashina2.2-3b-instruct-v0.1", "safeConfigVector": {"dtype": "-", "gpuNum": 1, "maxModelLen": 4096}, "status": "recovered_active", "targetGpu": "Sunrise_pt-200-x1", "taskId": "4986333", "taskType": "text-generation"}
{"batchId": "627b79e82e5d44d8809a85cc9fbe8757", "completedAt": "2026-09-20T19:44:35.040138+00:00", "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configSource": "modelhub_live", "createdAt": "2026-09-20T19:37:07.061559+00:00", "framework": "vllm_tokenizer_patch", "intentId": "1fd4bb7349cf4e348b6ed2ad4707abf9", "lastModified": "2026-08-24T19:46:26+00:00", "modelAddress": "https://modelscope.cn/models/neuralmagic/starcoder2-15b-quantized.w8a8", "reason": null, "reconciledAt": "2026-09-20T19:47:50.162472+00:00", "repoId": "neuralmagic/starcoder2-15b-quantized.w8a8", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "recovered_active", "targetGpu": "Ascend_910-b3", "taskId": "4986407", "taskType": "text-generation"}
{"batchId": "afdf88a0377042f18acdb6b1332d34a1", "completedAt": "2026-09-20T19:47:48.783546+00:00", "configFingerprint": "2e9cebc01b20081f0839fa18fe9ca4f7d059509f235699027201fe328177b304", "configSource": "modelhub_live", "createdAt": "2026-09-20T19:44:35.828352+00:00", "framework": "vllm_fix_tokenizer", "intentId": "3637e84970b84816b32fd346bd264346", "lastModified": "2026-08-26T18:08:39+00:00", "modelAddress": "https://modelscope.cn/models/aisingapore/Gemma-SEA-LION-v4-27B-IT", "reason": null, "reconciledAt": "2026-09-20T19:47:50.164764+00:00", "repoId": "aisingapore/Gemma-SEA-LION-v4-27B-IT", "safeConfigVector": {"dtype": "-", "gpuNum": 1, "maxModelLen": 4096}, "status": "recovered_active", "targetGpu": "Sunrise_pt-200-x1", "taskId": "4986586", "taskType": "text-generation"}
{"batchId": "6252f4ac62854ed7ac0796e8fdb39dc5", "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configSource": "modelhub_live", "createdAt": "2026-09-20T19:48:06.229065+00:00", "framework": "vllm_fix_tokenizer", "intentId": "f80b0994063c4feb8eeabd68353634d6", "lastModified": "2026-09-14T20:26:18+00:00", "modelAddress": "https://modelscope.cn/models/prithivMLmods/CEERS-2112-14B-Instruct", "repoId": "prithivMLmods/CEERS-2112-14B-Instruct", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Biren_166m", "taskType": "text-generation"}
{"batchId": "6252f4ac62854ed7ac0796e8fdb39dc5", "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configSource": "modelhub_live", "createdAt": "2026-09-20T19:48:06.229169+00:00", "framework": "vllm_fix_tokenizer", "intentId": "991315d5ee284fa59b657cee3f9f0357", "lastModified": "2026-09-17T14:05:32+00:00", "modelAddress": "https://modelscope.cn/models/Youssofal/Qwen3.8-27B-MTPLX-Optimized-Speed", "repoId": "Youssofal/Qwen3.8-27B-MTPLX-Optimized-Speed", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Biren_166m", "taskType": "text-generation"}
{"batchId": "6252f4ac62854ed7ac0796e8fdb39dc5", "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configSource": "modelhub_live", "createdAt": "2026-09-20T19:48:06.229217+00:00", "framework": "vllm_fix_tokenizer", "intentId": "603b57ae29cb4bcaac13f5f408f372e7", "lastModified": "2026-09-16T16:05:58+00:00", "modelAddress": "https://modelscope.cn/models/BAAI/RoboBrain2.5-8B-NV", "repoId": "BAAI/RoboBrain2.5-8B-NV", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Biren_166m", "taskType": "text-generation"}
{"batchId": "6252f4ac62854ed7ac0796e8fdb39dc5", "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configSource": "modelhub_live", "createdAt": "2026-09-20T19:48:06.229261+00:00", "framework": "vllm_fix_tokenizer", "intentId": "4d9d587be87d44229e3483926baf2e75", "lastModified": "2026-09-14T19:52:04+00:00", "modelAddress": "https://modelscope.cn/models/mlx-community/gemma-4-e4b-it-OptiQ-4bit", "repoId": "mlx-community/gemma-4-e4b-it-OptiQ-4bit", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Biren_166m", "taskType": "text-generation"}
{"batchId": "6252f4ac62854ed7ac0796e8fdb39dc5", "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configSource": "modelhub_live", "createdAt": "2026-09-20T19:48:06.229304+00:00", "framework": "vllm_fix_tokenizer", "intentId": "ea2a561cd50c482094c76a48aa49b4c0", "lastModified": "2026-09-17T12:51:11+00:00", "modelAddress": "https://modelscope.cn/models/Youssofal/Qwen3.8-27B-MTPLX-Optimized-Speed-FP16", "repoId": "Youssofal/Qwen3.8-27B-MTPLX-Optimized-Speed-FP16", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Biren_166m", "taskType": "text-generation"}
{"batchId": "6252f4ac62854ed7ac0796e8fdb39dc5", "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configSource": "modelhub_live", "createdAt": "2026-09-20T19:48:06.229347+00:00", "framework": "vllm_fix_tokenizer", "intentId": "6a87c98448634f39b800d44ca18bbec0", "lastModified": "2026-09-20T14:38:08+00:00", "modelAddress": "https://modelscope.cn/models/solidrust/Llama-3-8B-Instruct-v0.4-AWQ", "repoId": "solidrust/Llama-3-8B-Instruct-v0.4-AWQ", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Biren_166m", "taskType": "text-generation"}
{"batchId": "6252f4ac62854ed7ac0796e8fdb39dc5", "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configSource": "modelhub_live", "createdAt": "2026-09-20T19:48:06.229390+00:00", "framework": "vllm_fix_tokenizer", "intentId": "4bba68205323458bb91f4f99c5b0cdb2", "lastModified": "2026-09-20T14:39:27+00:00", "modelAddress": "https://modelscope.cn/models/solidrust/Llama-3-16B-Instruct-v0.1-AWQ", "repoId": "solidrust/Llama-3-16B-Instruct-v0.1-AWQ", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Biren_166m", "taskType": "text-generation"}
{"batchId": "6252f4ac62854ed7ac0796e8fdb39dc5", "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configSource": "modelhub_live", "createdAt": "2026-09-20T19:48:06.229433+00:00", "framework": "vllm_fix_tokenizer", "intentId": "b850115047d047d084320cae7891607a", "lastModified": "2026-09-20T14:21:01+00:00", "modelAddress": "https://modelscope.cn/models/solidrust/Llama-3-8B-Instruct-DPO-v0.1-AWQ", "repoId": "solidrust/Llama-3-8B-Instruct-DPO-v0.1-AWQ", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Biren_166m", "taskType": "text-generation"}
{"batchId": "6252f4ac62854ed7ac0796e8fdb39dc5", "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configSource": "modelhub_live", "createdAt": "2026-09-20T19:48:06.229476+00:00", "framework": "vllm_fix_tokenizer", "intentId": "45af77873ef943899bd9d9758543472a", "lastModified": "2026-09-20T14:29:29+00:00", "modelAddress": "https://modelscope.cn/models/solidrust/Llama-3-8B-Instruct-v0.5-AWQ", "repoId": "solidrust/Llama-3-8B-Instruct-v0.5-AWQ", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Biren_166m", "taskType": "text-generation"}
{"batchId": "6252f4ac62854ed7ac0796e8fdb39dc5", "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configSource": "modelhub_live", "createdAt": "2026-09-20T19:48:06.229518+00:00", "framework": "vllm_fix_tokenizer", "intentId": "10e59121ff154e91a946c44e9a0fcba6", "lastModified": "2026-09-17T13:39:53+00:00", "modelAddress": "https://modelscope.cn/models/Youssofal/Qwen3.8-27B-MTPLX-Bare-Speed", "repoId": "Youssofal/Qwen3.8-27B-MTPLX-Bare-Speed", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Biren_166m", "taskType": "text-generation"}
{"batchId": "6252f4ac62854ed7ac0796e8fdb39dc5", "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configSource": "modelhub_live", "createdAt": "2026-09-20T19:48:06.229561+00:00", "framework": "vllm_fix_tokenizer", "intentId": "9dd1093ba6a84dc5818800d603cb9fb1", "lastModified": "2026-09-20T14:50:03+00:00", "modelAddress": "https://modelscope.cn/models/solidrust/Llama-3-11B-Instruct-v0.1-AWQ", "repoId": "solidrust/Llama-3-11B-Instruct-v0.1-AWQ", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Biren_166m", "taskType": "text-generation"}
{"batchId": "6252f4ac62854ed7ac0796e8fdb39dc5", "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configSource": "modelhub_live", "createdAt": "2026-09-20T19:48:06.229604+00:00", "framework": "vllm_fix_tokenizer", "intentId": "30b1dae13c854984a24604f5d853db9e", "lastModified": "2026-09-15T14:59:20+00:00", "modelAddress": "https://modelscope.cn/models/mlx-community/gemma-4-12B-it-qat-OptiQ-4bit", "repoId": "mlx-community/gemma-4-12B-it-qat-OptiQ-4bit", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Biren_166m", "taskType": "text-generation"}
{"batchId": "6252f4ac62854ed7ac0796e8fdb39dc5", "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configSource": "modelhub_live", "createdAt": "2026-09-20T19:48:06.229647+00:00", "framework": "vllm_fix_tokenizer", "intentId": "8cb98008d2c24aceb7ac5ffa7bd6ab60", "lastModified": "2026-09-20T14:39:19+00:00", "modelAddress": "https://modelscope.cn/models/solidrust/Llama-3-13B-Instruct-v0.1-AWQ", "repoId": "solidrust/Llama-3-13B-Instruct-v0.1-AWQ", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Biren_166m", "taskType": "text-generation"}
{"batchId": "6252f4ac62854ed7ac0796e8fdb39dc5", "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configSource": "modelhub_live", "createdAt": "2026-09-20T19:48:06.229782+00:00", "framework": "vllm_fix_tokenizer", "intentId": "7d556d131f314bc89810674be74badf4", "lastModified": "2026-09-14T19:46:18+00:00", "modelAddress": "https://modelscope.cn/models/mlx-community/gemma-4-e2b-it-OptiQ-4bit", "repoId": "mlx-community/gemma-4-e2b-it-OptiQ-4bit", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Biren_166m", "taskType": "text-generation"}
{"batchId": "6252f4ac62854ed7ac0796e8fdb39dc5", "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configSource": "modelhub_live", "createdAt": "2026-09-20T19:48:06.229827+00:00", "framework": "vllm_fix_tokenizer", "intentId": "8a920204f48744b1891ad0c95d2f16d5", "lastModified": "2026-09-15T15:10:12+00:00", "modelAddress": "https://modelscope.cn/models/mlx-community/XYZ-Aquila-mini-OptiQ-4bit", "repoId": "mlx-community/XYZ-Aquila-mini-OptiQ-4bit", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Biren_166m", "taskType": "text-generation"}
{"batchId": "6252f4ac62854ed7ac0796e8fdb39dc5", "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configSource": "modelhub_live", "createdAt": "2026-09-20T19:48:06.229870+00:00", "framework": "vllm_fix_tokenizer", "intentId": "892463b5a54148beb85eac105d093132", "lastModified": "2026-09-17T10:58:53+00:00", "modelAddress": "https://modelscope.cn/models/SyonLi/Qwen3-4B-Instruct-2507-Segmenter", "repoId": "SyonLi/Qwen3-4B-Instruct-2507-Segmenter", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Biren_166m", "taskType": "text-generation"}
{"batchId": "6252f4ac62854ed7ac0796e8fdb39dc5", "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configSource": "modelhub_live", "createdAt": "2026-09-20T19:48:06.229913+00:00", "framework": "vllm_fix_tokenizer", "intentId": "c97630c25af641b897b4a8aa41373360", "lastModified": "2026-09-20T15:32:35+00:00", "modelAddress": "https://modelscope.cn/models/Arain119/Sophia", "repoId": "Arain119/Sophia", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Biren_166m", "taskType": "text-generation"}
{"batchId": "6252f4ac62854ed7ac0796e8fdb39dc5", "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configSource": "modelhub_live", "createdAt": "2026-09-20T19:48:06.229956+00:00", "framework": "vllm_fix_tokenizer", "intentId": "2272b724bcac45e1976fddd01b71d439", "lastModified": "2026-09-15T15:08:08+00:00", "modelAddress": "https://modelscope.cn/models/mlx-community/KAT-Coder-V2.5-Dev-OptiQ-4bit", "repoId": "mlx-community/KAT-Coder-V2.5-Dev-OptiQ-4bit", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Biren_166m", "taskType": "text-generation"}
{"batchId": "6252f4ac62854ed7ac0796e8fdb39dc5", "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configSource": "modelhub_live", "createdAt": "2026-09-20T19:48:06.229999+00:00", "framework": "vllm_fix_tokenizer", "intentId": "e37b8803d7744f07b5f6e6e984568166", "lastModified": "2026-09-17T15:15:04+00:00", "modelAddress": "https://modelscope.cn/models/mlx-community/gemma-4-31B-it-qat-OptiQ-4bit", "repoId": "mlx-community/gemma-4-31B-it-qat-OptiQ-4bit", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Biren_166m", "taskType": "text-generation"}
{"batchId": "6252f4ac62854ed7ac0796e8fdb39dc5", "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configSource": "modelhub_live", "createdAt": "2026-09-20T19:48:06.230041+00:00", "framework": "vllm_fix_tokenizer", "intentId": "be2aaf7c9b5642eaa37d7274500df3a5", "lastModified": "2026-09-15T15:01:23+00:00", "modelAddress": "https://modelscope.cn/models/mlx-community/gemma-4-26B-A4B-it-qat-OptiQ-4bit", "repoId": "mlx-community/gemma-4-26B-A4B-it-qat-OptiQ-4bit", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Biren_166m", "taskType": "text-generation"}
{"batchId": "6252f4ac62854ed7ac0796e8fdb39dc5", "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configSource": "modelhub_live", "createdAt": "2026-09-20T19:48:06.230084+00:00", "framework": "vllm_fix_tokenizer", "intentId": "6fac8b49bdbd4be6be4c6eeff75df012", "lastModified": "2026-09-15T14:57:59+00:00", "modelAddress": "https://modelscope.cn/models/mlx-community/Qwen3.5-122B-A10B-OptiQ-2bit-REAP-63B", "repoId": "mlx-community/Qwen3.5-122B-A10B-OptiQ-2bit-REAP-63B", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Biren_166m", "taskType": "text-generation"}
{"batchId": "6252f4ac62854ed7ac0796e8fdb39dc5", "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configSource": "modelhub_live", "createdAt": "2026-09-20T19:48:06.230128+00:00", "framework": "vllm_fix_tokenizer", "intentId": "7194cc1d1472499f9e6c92282dd605a0", "lastModified": "2026-09-17T13:54:18+00:00", "modelAddress": "https://modelscope.cn/models/Youssofal/Qwen3.8-27B-MTPLX-Bare-Speed-FP16", "repoId": "Youssofal/Qwen3.8-27B-MTPLX-Bare-Speed-FP16", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Biren_166m", "taskType": "text-generation"}
{"batchId": "6252f4ac62854ed7ac0796e8fdb39dc5", "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configSource": "modelhub_live", "createdAt": "2026-09-20T19:48:06.230170+00:00", "framework": "vllm_fix_tokenizer", "intentId": "53a89da44d404bc9b7cbd57caf79af10", "lastModified": "2026-09-16T17:02:01+00:00", "modelAddress": "https://modelscope.cn/models/mlx-community/Devstral-Small-2-24B-Instruct-2512-OptiQ-4bit", "repoId": "mlx-community/Devstral-Small-2-24B-Instruct-2512-OptiQ-4bit", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Biren_166m", "taskType": "text-generation"}
{"batchId": "6252f4ac62854ed7ac0796e8fdb39dc5", "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configSource": "modelhub_live", "createdAt": "2026-09-20T19:48:06.230213+00:00", "framework": "vllm_fix_tokenizer", "intentId": "2dd2cfc1733046af9ecbb358c073b383", "lastModified": "2026-09-16T17:13:44+00:00", "modelAddress": "https://modelscope.cn/models/mlx-community/LFM2.5-1.2B-Instruct-4bit", "repoId": "mlx-community/LFM2.5-1.2B-Instruct-4bit", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Biren_166m", "taskType": "text-generation"}
{"batchId": "6252f4ac62854ed7ac0796e8fdb39dc5", "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configSource": "modelhub_live", "createdAt": "2026-09-20T19:48:06.230256+00:00", "framework": "vllm_fix_tokenizer", "intentId": "37ebc03812d144f4adddeefb147baed5", "lastModified": "2026-09-17T13:29:50+00:00", "modelAddress": "https://modelscope.cn/models/Youssofal/Qwen3.5-9B-MTPLX-Optimized-Speed", "repoId": "Youssofal/Qwen3.5-9B-MTPLX-Optimized-Speed", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Biren_166m", "taskType": "text-generation"}
{"batchId": "6252f4ac62854ed7ac0796e8fdb39dc5", "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configSource": "modelhub_live", "createdAt": "2026-09-20T19:48:06.230299+00:00", "framework": "vllm_fix_tokenizer", "intentId": "550270da792c442d8df436de01029c61", "lastModified": "2026-09-15T15:11:14+00:00", "modelAddress": "https://modelscope.cn/models/mlx-community/gemma-4-e2b-it-qat-OptiQ-4bit", "repoId": "mlx-community/gemma-4-e2b-it-qat-OptiQ-4bit", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Biren_166m", "taskType": "text-generation"}
{"batchId": "afdf88a0377042f18acdb6b1332d34a1", "completedAt": "2026-09-20T19:47:48.783554+00:00", "configFingerprint": "2e9cebc01b20081f0839fa18fe9ca4f7d059509f235699027201fe328177b304", "configSource": "modelhub_live", "createdAt": "2026-09-20T19:44:35.828518+00:00", "framework": "vllm_fix_tokenizer", "intentId": "303c54082d6f4de78e21aaa05e4e8220", "repoId": "LiquidAI/LFM2.5-1.2B-Instruct-MLX-bf16", "safeConfigVector": {"dtype": "-", "gpuNum": 1, "maxModelLen": 4096}, "status": "age_policy_deferred", "targetGpu": "Sunrise_pt-200-x1", "taskId": null, "taskType": "text-generation"}
{"batchId": "afdf88a0377042f18acdb6b1332d34a1", "completedAt": "2026-09-20T19:47:48.783551+00:00", "configFingerprint": "2e9cebc01b20081f0839fa18fe9ca4f7d059509f235699027201fe328177b304", "configSource": "modelhub_live", "createdAt": "2026-09-20T19:44:35.828462+00:00", "framework": "vllm_fix_tokenizer", "intentId": "e8411decd0484162be84ed647c5bc79b", "repoId": "mlx-community/Spark-X2.5-4B-OptiQ-4bit", "safeConfigVector": {"dtype": "-", "gpuNum": 1, "maxModelLen": 4096}, "status": "age_policy_deferred", "targetGpu": "Sunrise_pt-200-x1", "taskId": null, "taskType": "text-generation"}
{"batchId": "afdf88a0377042f18acdb6b1332d34a1", "completedAt": "2026-09-20T19:47:48.783549+00:00", "configFingerprint": "2e9cebc01b20081f0839fa18fe9ca4f7d059509f235699027201fe328177b304", "configSource": "modelhub_live", "createdAt": "2026-09-20T19:44:35.828407+00:00", "framework": "vllm_fix_tokenizer", "intentId": "0b402cd014524ef0ba0bc89f78f9c821", "repoId": "aisingapore/Llama-SEA-LION-v3-8B", "safeConfigVector": {"dtype": "-", "gpuNum": 1, "maxModelLen": 4096}, "status": "age_policy_deferred", "targetGpu": "Sunrise_pt-200-x1", "taskId": null, "taskType": "text-generation"}

View File

@@ -1,24 +1,24 @@
{
"agentVersion": "2026.09.20.2",
"checksums": {
".modelhub_state/architecture_compatibility_blacklist.json": "c894bb05dc3ab263dd6dc0268cb61ff3fdaf8dc89b2b428155453619bb706384",
".modelhub_state/architecture_compatibility_blacklist.json": "c0375913c7866f56b0e21256cd1e562785ba355f661929bbe268cb6fb30acfa2",
".modelhub_state/architecture_history_backfill.json": "a92348206605da04f75c9e26e7c818b76f259e0b28983f18b6375ef94802f0ab",
".modelhub_state/market_intelligence.json": "9477690568c3884c93fa81cfad8e9500ff3009698b7e5578177383e813241b3f",
".modelhub_state/official_capabilities.json": "45ef891264b86427d5fa0ec64f7c83d341749a090ca72ebe56728050f4fddef6",
".modelhub_state/outcome_checkpoint.json": "641402cdc01fee0f7706e286dbf09c8e49222965806808f84e0f1309114a36c1",
".modelhub_state/market_intelligence.json": "3e2194611eca1c98fd6e4d459d3993bad0f42455acd5b502400f636a191f915d",
".modelhub_state/official_capabilities.json": "db6c29fe92ddc9d55e04cd5f42a2e6462d65280bd8801ea7287601f01e9973dd",
".modelhub_state/outcome_checkpoint.json": "64d2ee89754adf8885549310123cdedc06b885b09a0f2ea8cc04fce770994054",
".modelhub_state/queue_cleanup_latest.json": "51060760318808c95c5f937aac9389cef4379d95a33f6213764b0f798b1c6bb9",
".modelhub_state/recent_outcomes.jsonl": "bca9f8c48e8297165b151979e7aa5bfe8599e5d21a777f6afdb66eac2b2eb59f",
".modelhub_state/recent_outcomes.jsonl": "f394f77eaff8b342a5cc3752aeb626cc40c67ebb881842f93954228dc015a15c",
".modelhub_state/recovery_active_tasks.jsonl": "db0594b4f6cac381cf7845c3047272c8428a8df9382a26021c216d47c0a10897",
".modelhub_state/recovery_intents.jsonl": "620e8be013d5198256a692f3d884bb36ef6f7eb44e669a220cc1e29bfb4d752a",
".modelhub_state/recovery_intents.jsonl": "54982d0b39338e3b61031616ef35812406a5db37ca95ebf99f340247c3da049f",
".modelhub_state/routing_intelligence.json": "f4d5b8b9dcb1c204c6e1e30971a11d9cc28844fec061d5b484f538c84028653c",
".modelhub_state/submission_exclusions.jsonl": "221be895a524330815b3c5124450b33c94b5f1e68169030c73684bf675d06ec7",
".modelhub_state/worker_crashes.jsonl": "a494412048eca6dce83e7284303a6b4b56a445c1274ac201ab8723c739a14f23",
"ledger/submissions.jsonl": "9e95f17de83962dfc513d832d7bbcf17f7092ff95ecd233437f99450f89c1aa8",
"outcomes/submissions.jsonl": "a0a3610707423e38c45c70a7c0314d07d919d65b4f9a9a68412797e983aabc39"
"outcomes/submissions.jsonl": "57412b23adc88734193f2ef77bb3041c569eec6af94f635bbd9bf92d0ec385e2"
},
"generation": 10837,
"phase": "cycle",
"generation": 10838,
"phase": "intent",
"schemaVersion": 1,
"updatedAt": "2026-09-20T19:47:50.328226+00:00",
"updatedAt": "2026-09-20T19:48:06.386167+00:00",
"writerId": "fc715c06e24c4db2ae3e425e435db996"
}

View File

@@ -509,7 +509,6 @@
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T11:00:25.647313+00:00", "modelId": "JANGQ-AI/Osaurus-AppleScript-8B-JANG_4M", "modelProfile": {"architectures": ["ZayaForCausalLM"], "configFingerprint": "8b3c0a82b858cf26b1a0f6075074234a66487d5734741dbbfd4743963b967faa", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7915599888, "estimatedRequiredGiB": 8.885, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "zaya", "modelscopeFileSize": 7950463599, "modelscopeLicense": "other", "modelscopeParams": 2274126328, "modelscopeTags": ["license:other", "model_type:zaya", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:applescript", "custom_tag:macos", "custom_tag:computer-use", "custom_tag:agent", "custom_tag:tool-calling", "custom_tag:function-calling", "custom_tag:mlx", "custom_tag:moe", "custom_tag:osaurus"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 7950463599}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T02:52:52.242161+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4969087", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T11:00:25.647652+00:00", "modelId": "aisingapore/Llama-SEA-LION-v3.5-8B-R", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "8b3c0a82b858cf26b1a0f6075074234a66487d5734741dbbfd4743963b967faa", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 16060556344, "estimatedRequiredGiB": 17.97, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 16079585018, "modelscopeLicense": "llama3.1", "modelscopeParams": 8030261248, "modelscopeTags": ["license:llama3.1", "model_type:llama", "library:pytorch", "library:safetensors", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 16079585018}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T02:52:52.306108+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4969091", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T11:00:25.647389+00:00", "modelId": "ysqlian/YZH_Model", "modelProfile": {"architectures": ["Qwen2ForCausalLM"], "configFingerprint": "8b3c0a82b858cf26b1a0f6075074234a66487d5734741dbbfd4743963b967faa", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 6171927000, "estimatedRequiredGiB": 6.915, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen2", "modelscopeFileSize": 6187845052, "modelscopeLicense": "Apache License 2.0", "modelscopeParams": 3085938688, "modelscopeTags": ["license:Apache License 2.0", "model_type:qwen2", "library:safetensors", "library:other", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 6187845052}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T02:52:52.308078+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4969092", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": null, "modelId": "dickeyli/llmrec-onereason-8b-sh2-grpo4-step160", "modelProfile": {"architectures": ["Qwen3ForCausalLM"], "configFingerprint": "8b3c0a82b858cf26b1a0f6075074234a66487d5734741dbbfd4743963b967faa", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 16779959304, "estimatedRequiredGiB": 18.777, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3", "modelscopeFileSize": 16801630203, "modelscopeLicense": "Apache License 2.0", "modelscopeParams": 8389956608, "modelscopeTags": ["license:Apache License 2.0", "model_type:qwen3", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 16801630203}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-20T02:52:52.295134+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4969090", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T11:00:25.647526+00:00", "modelId": "ibm-granite/granite-guardian-4.1-8b", "modelProfile": {"architectures": ["GraniteForCausalLM"], "configFingerprint": "a494b1667fab53609b94cc853db1a12c8d3b9fcf98795ecbab3dd5eb3a3c77f6", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 16761144464, "estimatedRequiredGiB": 18.743, "gpuMemoryEvidence": {"memoryGiB": 48.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 48.0, "maximumRepositorySizeGiB": 40.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "granite", "modelscopeFileSize": 16770924251, "modelscopeLicense": "apache-2.0", "modelscopeParams": 8380551168, "modelscopeTags": ["license:apache-2.0", "model_type:granite", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:granite", "custom_tag:guardian", "custom_tag:safety", "custom_tag:hallucination"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 16770924251}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T02:52:52.310071+00:00", "targetGpu": "Mthreads_s4000", "taskId": "4969093", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T11:10:38.766271+00:00", "modelId": "neuralmagic/Mistral-Nemo-Instruct-2407-quantized.w4a16", "modelProfile": {"architectures": ["MistralForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 8308282608, "estimatedRequiredGiB": 9.319, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "mistral", "modelscopeFileSize": 8338221588, "modelscopeLicense": "llama2", "modelscopeParams": 12247782400, "modelscopeTags": ["license:llama2", "model_type:mistral", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 8338221588}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:09:11.490200+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4969296", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T11:10:38.766176+00:00", "modelId": "LiquidAI/LFM2.5-8B-A1B", "modelProfile": {"architectures": ["Lfm2MoeForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 16936006912, "estimatedRequiredGiB": 18.948, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2_moe", "modelscopeFileSize": 16953948786, "modelscopeLicense": "other", "modelscopeParams": 8467856832, "modelscopeTags": ["license:other", "model_type:lfm2_moe", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 16953948786}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:09:11.550122+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4969304", "taskType": "text-generation", "verifyResult": null}
@@ -839,7 +838,6 @@
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-20T12:21:49.901954+00:00", "modelId": "mlx-community/gemma-4-e2b-it-OptiQ-4bit", "modelProfile": {"architectures": ["Gemma4ForConditionalGeneration"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 5226595850, "estimatedRequiredGiB": 5.878, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma4", "modelscopeFileSize": 5259113427, "modelscopeLicense": "gemma", "modelscopeParams": 1141165347, "modelscopeTags": ["license:gemma", "model_type:gemma4", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:quantized", "custom_tag:mixed-precision", "custom_tag:4bit", "custom_tag:8bit", "custom_tag:optiq", "custom_tag:apple-silicon", "custom_tag:text-generation", "custom_tag:gemma-4"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 5259113427}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:09:56.538457+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4970504", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-20T12:21:49.901949+00:00", "modelId": "mlx-community/Devstral-Small-2-24B-Instruct-2512-OptiQ-4bit", "modelProfile": {"architectures": ["Mistral3ForConditionalGeneration"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 17367755740, "estimatedRequiredGiB": 19.429, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "mistral3", "modelscopeFileSize": 17385072379, "modelscopeLicense": "apache-2.0", "modelscopeParams": 4929899520, "modelscopeTags": ["license:apache-2.0", "model_type:mistral3", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:quantized", "custom_tag:mixed-precision", "custom_tag:4bit", "custom_tag:8bit", "custom_tag:optiq", "custom_tag:apple-silicon", "custom_tag:text-generation", "custom_tag:mistral", "custom_tag:mistral3", "custom_tag:devstral"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 17385072379}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:09:56.486238+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4970511", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-mlu", "lastSyncTime": "2026-09-20T12:21:49.901911+00:00", "modelId": "pfnet/plamo-3-nict-8b-base", "modelProfile": {"architectures": ["Plamo3ForCausalLM"], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 16182734928, "estimatedRequiredGiB": 18.09, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "plamo3", "modelscopeFileSize": 16186391993, "modelscopeLicense": "other", "modelscopeParams": 8091348992, "modelscopeTags": ["license:other", "model_type:plamo3", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 16186391993}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:09:56.487931+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4970506", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": null, "modelId": "aisingapore/SEA-LION-v1-7B-IT-GPTQ", "modelProfile": {"architectures": ["MPTForCausalLM"], "configFingerprint": "f7d2aee0bcf1396cd8ab3b1064b6a032eccc04f2d655a336023e13528574903d", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 5469228368, "estimatedRequiredGiB": 6.118, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "mpt", "modelscopeFileSize": 5473985594, "modelscopeLicense": "mit", "modelscopeParams": 1922048000, "modelscopeTags": ["license:mit", "model_type:mpt", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 5473985594}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-20T04:09:56.541899+00:00", "targetGpu": "Biren_166m", "taskId": "4970509", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T12:21:49.902104+00:00", "modelId": "LiquidAI/LFM2.5-8B-A1B", "modelProfile": {"architectures": ["Lfm2MoeForCausalLM"], "configFingerprint": "6db01b5037d5ed2db74875d2113f8582c97f34bd825637161e54174c9f2cabed", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 16936006912, "estimatedRequiredGiB": 18.948, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2_moe", "modelscopeFileSize": 16953948786, "modelscopeLicense": "other", "modelscopeParams": 8467856832, "modelscopeTags": ["license:other", "model_type:lfm2_moe", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 16953948786}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:09:56.546014+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4970502", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-customized", "lastSyncTime": "2026-09-20T12:21:49.901738+00:00", "modelId": "LiquidAI/LFM2.5-1.2B-Thinking-MLX-4bit", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 658540250, "estimatedRequiredGiB": 0.741, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 663410366, "modelscopeLicense": "other", "modelscopeParams": 182975232, "modelscopeTags": ["license:other", "model_type:lfm2", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge", "custom_tag:mlx", "custom_tag:reasoning"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 663410366}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:10:03.083968+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4970513", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T12:21:49.901858+00:00", "modelId": "inceptionai/Jais-2-8B-Chat", "modelProfile": {"architectures": ["Jais2ForCausalLM"], "configFingerprint": "533968aac3b9c90272dfbe2866791f782e077f0dceee5b6ffffce48ba6674a10", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 16180861064, "estimatedRequiredGiB": 18.097, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "jais2", "modelscopeFileSize": 16193187697, "modelscopeLicense": "apache-2.0", "modelscopeParams": 8090401280, "modelscopeTags": ["license:apache-2.0", "model_type:jais2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 16193187697}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:10:03.058495+00:00", "targetGpu": "MetaX_c-500", "taskId": "4970512", "taskType": "text-generation", "verifyResult": null}
@@ -904,7 +902,6 @@
{"failReason": null, "framework": "vllm-customized", "lastSyncTime": "2026-09-20T13:34:22.859841+00:00", "modelId": "Youssofal/Qwen3.8-27B-MTPLX-Bare-Speed-FP16", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 16293473597, "estimatedRequiredGiB": 18.233, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 16314185171, "modelscopeLicense": "apache-2.0", "modelscopeParams": 4665462000, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:apple-silicon", "custom_tag:macos", "custom_tag:m1", "custom_tag:m2", "custom_tag:fp16", "custom_tag:speculative-decoding", "custom_tag:multi-token-prediction", "custom_tag:qwen", "custom_tag:qwen3.8", "custom_tag:mtp", "custom_tag:mtplx", "custom_tag:local-ai", "custom_tag:chat", "custom_tag:qwen3-8", "custom_tag:qwen-3.8", "custom_tag:local-llm", "custom_tag:llm", "custom_tag:m5", "custom_tag:m5-max", "custom_tag:m4", "custom_tag:m3", "custom_tag:macbook-pro", "custom_tag:mac-studio", "custom_tag:opencode", "custom_tag:claude-code", "custom_tag:27b", "custom_tag:qwen3.8-27b", "custom_tag:qwen3-8-27b", "custom_tag:coding"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 16314185171}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T05:31:42.867294+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4971507", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T14:18:03.259019+00:00", "modelId": "RedHatAI/gemma-2-2b-quantized.w8a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 6746735136, "estimatedRequiredGiB": 7.565, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 6768615358, "modelscopeLicense": "gemma", "modelscopeParams": 3204165888, "modelscopeTags": ["license:gemma", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 6768615358}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T06:16:07.298594+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4972201", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-20T14:18:03.258951+00:00", "modelId": "RedHatAI/gemma-2-27b-it-quantized.w8a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 30775446656, "estimatedRequiredGiB": 34.419, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 30797371689, "modelscopeLicense": "gemma", "modelscopeParams": 28406776320, "modelscopeTags": ["license:gemma", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 30797371689}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T06:16:07.269550+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4972172", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": null, "modelId": "neuralmagic/Qwen2-0.5B-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["Qwen2ForCausalLM"], "configFingerprint": "d341b8002cf39d3299fd429479e8c8e954583e49d4fec5a640ddc0f8f0cbd054", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 903168128, "estimatedRequiredGiB": 1.022, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen2", "modelscopeFileSize": 914658310, "modelscopeLicense": "apache-2.0", "modelscopeParams": 630167424, "modelscopeTags": ["license:apache-2.0", "model_type:qwen2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 914658310}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-20T06:17:57.834294+00:00", "targetGpu": "Biren_166m", "taskId": "4972230", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T14:32:15.260743+00:00", "modelId": "neuralmagic/Llama-3.2-1B-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2024670536, "estimatedRequiredGiB": 2.273, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 2033824885, "modelscopeLicense": "llama3.2", "modelscopeParams": 1498482688, "modelscopeTags": ["license:llama3.2", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:llama", "custom_tag:llama-3", "custom_tag:neuralmagic", "custom_tag:llmcompressor"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 2033824885}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T06:31:52.552038+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4972399", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T14:34:57.856700+00:00", "modelId": "neuralmagic/gemma-2-9b-it-quantized.w4a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7963207224, "estimatedRequiredGiB": 8.924, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 7985124714, "modelscopeLicense": "llama2", "modelscopeParams": 10159209984, "modelscopeTags": ["license:llama2", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 7985124714}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T06:32:07.434077+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4972417", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-mlu", "lastSyncTime": "2026-09-20T14:34:57.856737+00:00", "modelId": "RedHatAI/Phi-3-medium-128k-instruct-FP8", "modelProfile": {"architectures": ["Phi3ForCausalLM"], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 14289052464, "estimatedRequiredGiB": 15.972, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "phi3", "modelscopeFileSize": 14291553638, "modelscopeLicense": "mit", "modelscopeParams": 13960238080, "modelscopeTags": ["license:mit", "model_type:phi3", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:fp8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 14291553638}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T06:33:31.623471+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4972418", "taskType": "text-generation", "verifyResult": null}
@@ -956,8 +953,8 @@
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T19:33:34.552434+00:00", "modelId": "neuralmagic/starcoder2-15b-quantized.w8a8", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 17785240496, "estimatedRequiredGiB": 19.88, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 17788612744, "modelscopeLicense": "bigcode-openrail-m", "modelscopeParams": 15957889024, "modelscopeTags": ["license:bigcode-openrail-m", "model_type:starcoder2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 17788612744}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T11:19:51.535102+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4977812", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T19:33:34.552425+00:00", "modelId": "RedHatAI/Phi-3-medium-128k-instruct-quantized.w8a16", "modelProfile": {"architectures": ["Phi3ForCausalLM"], "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 14293356304, "estimatedRequiredGiB": 15.977, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "phi3", "modelscopeFileSize": 14295851911, "modelscopeLicense": "mit", "modelscopeParams": 13960238080, "modelscopeTags": ["license:mit", "model_type:phi3", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 14295851911}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T11:19:51.641136+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4977829", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T19:33:34.552441+00:00", "modelId": "RedHatAI/gemma-2-9b-it-quantized.w4a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7963207224, "estimatedRequiredGiB": 8.924, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 7985124576, "modelscopeLicense": "llama2", "modelscopeParams": 10159209984, "modelscopeTags": ["license:llama2", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 7985124576}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T11:20:03.373586+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4977833", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": null, "modelId": "neuralmagic/starcoder2-3b-quantized.w8a16", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4093180600, "estimatedRequiredGiB": 4.578, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 4096479112, "modelscopeLicense": "bigcode-openrail-m", "modelscopeParams": 3181366272, "modelscopeTags": ["license:bigcode-openrail-m", "model_type:starcoder2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 4096479112}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-20T11:39:31.087174+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4978074", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": null, "modelId": "neuralmagic/Meta-Llama-3.1-8B-Instruct-quantized.w8a16", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 9084040952, "estimatedRequiredGiB": 10.163, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 9093261938, "modelscopeLicense": "llama3.1", "modelscopeParams": 8030261248, "modelscopeTags": ["license:llama3.1", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 9093261938}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-20T11:39:31.092445+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4978075", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T19:47:52.978262+00:00", "modelId": "neuralmagic/starcoder2-3b-quantized.w8a16", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4093180600, "estimatedRequiredGiB": 4.578, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 4096479112, "modelscopeLicense": "bigcode-openrail-m", "modelscopeParams": 3181366272, "modelscopeTags": ["license:bigcode-openrail-m", "model_type:starcoder2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 4096479112}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T11:39:31.087174+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4978074", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-20T19:47:52.978218+00:00", "modelId": "neuralmagic/Meta-Llama-3.1-8B-Instruct-quantized.w8a16", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 9084040952, "estimatedRequiredGiB": 10.163, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 9093261938, "modelscopeLicense": "llama3.1", "modelscopeParams": 8030261248, "modelscopeTags": ["license:llama3.1", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 9093261938}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T11:39:31.092445+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4978075", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "transformers", "lastSyncTime": null, "modelId": "mlx-community/Qwen3.5-35B-A3B-OptiQ-4bit-REAP-19B", "modelProfile": {"architectures": ["Qwen3_5MoeForConditionalGeneration"], "configFingerprint": "3086a71adcdca90005b4967eb30749c44cd82389aff8d599758e4b91675d40dc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 13736540899, "estimatedRequiredGiB": 15.375, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_moe", "modelscopeFileSize": 13756990943, "modelscopeLicense": "apache-2.0", "modelscopeParams": 3825799024, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5_moe", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:quantized", "custom_tag:expert-pruning", "custom_tag:reap", "custom_tag:moe", "custom_tag:optiq", "custom_tag:apple-silicon", "custom_tag:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 13756990943}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-20T11:44:01.235722+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4978143", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": null, "modelId": "RedHatAI/starcoder2-3b-quantized.w8a16", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4093180600, "estimatedRequiredGiB": 4.578, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 4096479058, "modelscopeLicense": "bigcode-openrail-m", "modelscopeParams": 3181366272, "modelscopeTags": ["license:bigcode-openrail-m", "model_type:starcoder2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 4096479058}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-20T11:51:59.601108+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4978230", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": null, "modelId": "neuralmagic/Phi-3-medium-128k-instruct-quantized.w4a16", "modelProfile": {"architectures": ["Phi3ForCausalLM"], "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7686303632, "estimatedRequiredGiB": 8.592, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "phi3", "modelscopeFileSize": 7688299919, "modelscopeLicense": "llama2", "modelscopeParams": 13960238080, "modelscopeTags": ["license:llama2", "model_type:phi3", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 7688299919}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-20T11:52:06.534997+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4978249", "taskType": "text-generation", "verifyResult": null}