From 3b06c798feea82625bc7a5b592e7d1018c106e4f Mon Sep 17 00:00:00 2001 From: CoolBoy <2269097679@qq.com> Date: Sun, 20 Sep 2026 14:48:04 +0000 Subject: [PATCH] state: generation 10635 (intent) --- .../architecture_compatibility_blacklist.json | 2 +- .modelhub_state/market_intelligence.json | 6 +- .modelhub_state/official_capabilities.json | 131 +++++++-------- .modelhub_state/outcome_checkpoint.json | 158 +++++++++++++----- .modelhub_state/recent_outcomes.jsonl | 2 +- .modelhub_state/recovery_intents.jsonl | 1 + manifest.json | 20 +-- outcomes/submissions.jsonl | 4 +- 8 files changed, 191 insertions(+), 133 deletions(-) diff --git a/.modelhub_state/architecture_compatibility_blacklist.json b/.modelhub_state/architecture_compatibility_blacklist.json index ead66377b..75862cb14 100644 --- a/.modelhub_state/architecture_compatibility_blacklist.json +++ b/.modelhub_state/architecture_compatibility_blacklist.json @@ -1937,7 +1937,7 @@ "taskType": "text-generation" } }, - "generatedAt": "2026-09-20T14:34:58.383989+00:00", + "generatedAt": "2026-09-20T14:47:45.613208+00:00", "summary": { "activeBlockCount": 94, "byGpuFramework": { diff --git a/.modelhub_state/market_intelligence.json b/.modelhub_state/market_intelligence.json index 79e7c8948..9a9fbfc0c 100644 --- a/.modelhub_state/market_intelligence.json +++ b/.modelhub_state/market_intelligence.json @@ -1,8 +1,8 @@ { - "communityAttemptedAt": "2026-09-20T14:32:15.696895+00:00", + "communityAttemptedAt": "2026-09-20T14:47:53.539698+00:00", "communityError": null, "communitySample": {}, - "communityUpdatedAt": "2026-09-20T14:32:15.696895+00:00", + "communityUpdatedAt": "2026-09-20T14:47:53.539698+00:00", "frameworkAttemptedAt": "2026-09-20T11:15:54.454361+00:00", "frameworkError": null, "frameworkStats": { @@ -425,7 +425,7 @@ } }, "frameworkUpdatedAt": null, - "generatedAt": "2026-09-20T14:46:31.591216+00:00", + "generatedAt": "2026-09-20T14:47:53.539698+00:00", "gpuStats": { "Ascend_910-b3": { "available": true, diff --git a/.modelhub_state/official_capabilities.json b/.modelhub_state/official_capabilities.json index 8540edaf0..5ebfa108f 100644 --- a/.modelhub_state/official_capabilities.json +++ b/.modelhub_state/official_capabilities.json @@ -1,5 +1,5 @@ { - "catalogUpdatedAt": "2026-09-20T14:46:31.591216+00:00", + "catalogUpdatedAt": "2026-09-20T14:47:53.539698+00:00", "configuredTaskTypes": [ "text-generation" ], @@ -56,7 +56,7 @@ "time-series-forecasting" ], "errors": [], - "generatedAt": "2026-09-20T14:46:44.393462+00:00", + "generatedAt": "2026-09-20T14:48:03.352500+00:00", "gpuCatalog": { "Ascend_910-b3": { "canVerify": true, @@ -252,6 +252,12 @@ ], "updatedAt": "2026-09-20T14:37:05.364934+00:00" }, + "https://modelscope.cn/models/IntervitensInc/kek_mk3|2026-09-14T13:40:21+00:00|Cambricon_mlu-370-x8": { + "taskTypes": [ + "text-generation" + ], + "updatedAt": "2026-09-20T14:48:03.352500+00:00" + }, "https://modelscope.cn/models/JXW7777/TGAI_NB|2026-09-20T06:13:38+00:00|Ascend_910-b3": { "taskTypes": [ "text-generation", @@ -850,55 +856,6 @@ ], "updatedAt": "2026-09-20T14:46:34.773631+00:00" }, - "https://modelscope.cn/models/RedHatAI/GLM-5.2-speculator.dspark|2026-08-28T15:00:56+00:00|Biren_166m": { - "taskTypes": [ - "text-generation" - ], - "updatedAt": "2026-09-20T14:36:48.881874+00:00" - }, - "https://modelscope.cn/models/RedHatAI/GLM-5.2-speculator.dspark|2026-08-28T15:00:56+00:00|Cambricon_mlu-370-x4": { - "taskTypes": [ - "text-generation", - "text-to-image-generation", - "visual-multi-modal" - ], - "updatedAt": "2026-09-20T14:36:48.938062+00:00" - }, - "https://modelscope.cn/models/RedHatAI/GLM-5.2-speculator.dspark|2026-08-28T15:00:56+00:00|Cambricon_mlu-370-x8": { - "taskTypes": [ - "text-generation", - "text-to-image-generation", - "visual-multi-modal" - ], - "updatedAt": "2026-09-20T14:36:48.909414+00:00" - }, - "https://modelscope.cn/models/RedHatAI/GLM-5.2-speculator.dspark|2026-08-28T15:00:56+00:00|Iluvatar_bi-150": { - "taskTypes": [ - "asr", - "question_answering", - "reinforcement_learning", - "text-generation", - "text-to-image-generation", - "text_classification", - "vision_classification", - "visual-multi-modal" - ], - "updatedAt": "2026-09-20T14:36:48.966970+00:00" - }, - "https://modelscope.cn/models/RedHatAI/GLM-5.2-speculator.dspark|2026-08-28T15:00:56+00:00|Vastai_va16": { - "taskTypes": [ - "text-generation" - ], - "updatedAt": "2026-09-20T14:36:48.997532+00:00" - }, - "https://modelscope.cn/models/RedHatAI/GLM-5.2-speculator.dspark|2026-08-28T15:00:56+00:00|hygon_k100-ai": { - "taskTypes": [ - "text-generation", - "text-to-image-generation", - "visual-multi-modal" - ], - "updatedAt": "2026-09-20T14:36:49.025426+00:00" - }, "https://modelscope.cn/models/RedHatAI/Llama-2-7b-chat-quantized.w8a8|2026-08-24T19:25:45+00:00|Ascend_910-b4": { "taskTypes": [ "text-generation" @@ -2993,20 +2950,6 @@ ], "updatedAt": "2026-09-20T14:46:44.008290+00:00" }, - "https://modelscope.cn/models/guaidao2/XuanmuSec-2.6B|2026-08-28T06:49:12+00:00|Cambricon_mlu-370-x4": { - "taskTypes": [ - "text-generation", - "text-to-image-generation", - "visual-multi-modal" - ], - "updatedAt": "2026-09-20T14:36:48.816781+00:00" - }, - "https://modelscope.cn/models/guaidao2/XuanmuSec-2.6B|2026-08-28T06:49:12+00:00|Vastai_va16": { - "taskTypes": [ - "text-generation" - ], - "updatedAt": "2026-09-20T14:36:48.846921+00:00" - }, "https://modelscope.cn/models/icychick/Qwen3.5-text-0.8B-GGUF|2026-09-14T05:41:13+00:00|Ascend_910-b3": { "taskTypes": [ "text-generation", @@ -3264,6 +3207,14 @@ ], "updatedAt": "2026-09-20T14:46:38.913519+00:00" }, + "https://modelscope.cn/models/mlx-community/KAT-Coder-V2.5-Dev-OptiQ-4bit|2026-09-15T15:08:08+00:00|Ascend_910-b3": { + "taskTypes": [ + "text-generation", + "text-to-image-generation", + "visual-multi-modal" + ], + "updatedAt": "2026-09-20T14:48:03.071973+00:00" + }, "https://modelscope.cn/models/mlx-community/KAT-Coder-V2.5-Dev-OptiQ-4bit|2026-09-15T15:08:08+00:00|Biren_166m": { "taskTypes": [ "text-generation" @@ -3438,12 +3389,6 @@ ], "updatedAt": "2026-09-20T14:46:43.445951+00:00" }, - "https://modelscope.cn/models/mlx-community/Qwen3.6-35B-A3B-OptiQ-4bit-REAP-19B|2026-08-28T09:38:42+00:00|Biren_166m": { - "taskTypes": [ - "text-generation" - ], - "updatedAt": "2026-09-20T14:36:49.171372+00:00" - }, "https://modelscope.cn/models/mlx-community/Qwen3.6-35B-A3B-OptiQ-4bit-REAP-19B|2026-08-28T09:38:42+00:00|Iluvatar_bi-150": { "taskTypes": [ "asr", @@ -5164,6 +5109,24 @@ ], "updatedAt": "2026-09-20T14:46:34.121593+00:00" }, + "https://modelscope.cn/models/solidrust/Llama-3-13B-Instruct-v0.1-AWQ|2026-09-20T14:39:19+00:00|Ascend_910-b3": { + "taskTypes": [ + "text-generation" + ], + "updatedAt": "2026-09-20T14:47:59.037893+00:00" + }, + "https://modelscope.cn/models/solidrust/Llama-3-13B-Instruct-v0.1-AWQ|2026-09-20T14:39:19+00:00|Biren_166m": { + "taskTypes": [ + "text-generation" + ], + "updatedAt": "2026-09-20T14:47:59.147559+00:00" + }, + "https://modelscope.cn/models/solidrust/Llama-3-13B-Instruct-v0.1-AWQ|2026-09-20T14:39:19+00:00|Mthreads_s4000": { + "taskTypes": [ + "text-generation" + ], + "updatedAt": "2026-09-20T14:47:59.092054+00:00" + }, "https://modelscope.cn/models/solidrust/Llama-3-16B-Instruct-v0.1-AWQ|2026-09-19T14:40:27+00:00|Biren_166m": { "taskTypes": [ "text-generation" @@ -5176,6 +5139,18 @@ ], "updatedAt": "2026-09-20T14:46:33.890307+00:00" }, + "https://modelscope.cn/models/solidrust/Llama-3-16B-Instruct-v0.1-AWQ|2026-09-20T14:39:27+00:00|Biren_166m": { + "taskTypes": [ + "text-generation" + ], + "updatedAt": "2026-09-20T14:47:59.073292+00:00" + }, + "https://modelscope.cn/models/solidrust/Llama-3-16B-Instruct-v0.1-AWQ|2026-09-20T14:39:27+00:00|Mthreads_s4000": { + "taskTypes": [ + "text-generation" + ], + "updatedAt": "2026-09-20T14:47:59.036697+00:00" + }, "https://modelscope.cn/models/solidrust/Llama-3-8B-Instruct-DPO-v0.1-AWQ|2026-09-20T14:21:01+00:00|Biren_166m": { "taskTypes": [ "text-generation" @@ -5194,6 +5169,18 @@ ], "updatedAt": "2026-09-20T14:46:34.392926+00:00" }, + "https://modelscope.cn/models/solidrust/Llama-3-8B-Instruct-v0.4-AWQ|2026-09-20T14:38:08+00:00|Ascend_910-b3": { + "taskTypes": [ + "text-generation" + ], + "updatedAt": "2026-09-20T14:47:59.041942+00:00" + }, + "https://modelscope.cn/models/solidrust/Llama-3-8B-Instruct-v0.4-AWQ|2026-09-20T14:38:08+00:00|Biren_166m": { + "taskTypes": [ + "text-generation" + ], + "updatedAt": "2026-09-20T14:47:59.083921+00:00" + }, "https://modelscope.cn/models/solidrust/Llama-3-8B-Instruct-v0.5-AWQ|2026-09-20T14:29:29+00:00|Ascend_910-b3": { "taskTypes": [ "text-generation" @@ -6362,6 +6349,6 @@ "updateTime": "2025-12-22 08:59:53" } ], - "taskTreeUpdatedAt": "2026-09-20T14:46:31.591216+00:00", + "taskTreeUpdatedAt": "2026-09-20T14:47:53.539698+00:00", "version": 1 } diff --git a/.modelhub_state/outcome_checkpoint.json b/.modelhub_state/outcome_checkpoint.json index 13e5f9e41..58fa832f0 100644 --- a/.modelhub_state/outcome_checkpoint.json +++ b/.modelhub_state/outcome_checkpoint.json @@ -1,6 +1,6 @@ { - "generatedAt": "2026-09-20T14:34:58.328009+00:00", - "lastSyncTime": "2026-09-20T14:34:57.856745+00:00", + "generatedAt": "2026-09-20T14:47:45.558449+00:00", + "lastSyncTime": "2026-09-20T14:47:45.260018+00:00", "recentLimit": 300, "report": { "architectureCompatibilityBlocks": { @@ -3145,10 +3145,10 @@ "decisionSuccessRate": 0.0, "decisionTotal": 7, "failureBreakdown": { - "ambiguous_runtime": 24, + "ambiguous_runtime": 25, "memory_capacity": 7 }, - "failureCount": 31, + "failureCount": 32, "failureRate": 1.0, "framework": "vllm_fix_tokenizer", "pendingCount": 0, @@ -3158,8 +3158,8 @@ "successRate": 0.0, "targetGpu": "Iluvatar_bi-150", "taskType": "text-generation", - "total": 31, - "unresolvedFailureCount": 24 + "total": 32, + "unresolvedFailureCount": 25 }, "Iluvatar_bi-150|vllm|reinforcement_learning": { "attributableFailureCount": 0, @@ -3484,10 +3484,10 @@ "decisionSuccessRate": 0.0, "decisionTotal": 0, "failureBreakdown": { - "ambiguous_runtime": 82, + "ambiguous_runtime": 83, "参数/模板问题": 2 }, - "failureCount": 84, + "failureCount": 85, "failureRate": 1.0, "framework": "vllm_fix_tokenizer", "pendingCount": 0, @@ -3497,8 +3497,8 @@ "successRate": 0.0, "targetGpu": "Kunlunxin_p-800", "taskType": "text-generation", - "total": 84, - "unresolvedFailureCount": 84 + "total": 85, + "unresolvedFailureCount": 85 }, "Kunlunxin_r-200-8f|unknown|text-generation": { "attributableFailureCount": 0, @@ -4575,7 +4575,7 @@ "decisionSuccessRate": 0.0167, "decisionTotal": 120, "failureBreakdown": { - "ambiguous_runtime": 117, + "ambiguous_runtime": 119, "backend_operator": 8, "framework_architecture_unsupported": 20, "memory_capacity": 7, @@ -4584,15 +4584,15 @@ "tokenizer_compatibility": 82, "参数/模板问题": 8 }, - "failureCount": 259, - "failureRate": 0.9923, + "failureCount": 261, + "failureRate": 0.9924, "pendingCount": 0, "pendingRate": 0.0, "platformFailureCount": 16, "successCount": 2, - "successRate": 0.0077, - "total": 261, - "unresolvedFailureCount": 125 + "successRate": 0.0076, + "total": 263, + "unresolvedFailureCount": 127 }, "vllm_tokenizer_patch": { "attributableFailureCount": 26, @@ -4614,7 +4614,7 @@ "unresolvedFailureCount": 24 } }, - "generatedAt": "2026-09-20T14:34:58.319961+00:00", + "generatedAt": "2026-09-20T14:47:45.550467+00:00", "gpuSummaries": { "Ascend_910-b3": { "attributableFailureCount": 75, @@ -4781,7 +4781,7 @@ "decisionSuccessRate": 0.1701, "decisionTotal": 876, "failureBreakdown": { - "ambiguous_runtime": 604, + "ambiguous_runtime": 605, "architecture_compatibility": 15, "backend_operator": 11, "context_length": 29, @@ -4796,15 +4796,15 @@ "日志缺失": 155, "验证失败": 30 }, - "failureCount": 1856, + "failureCount": 1857, "failureRate": 0.9257, "pendingCount": 0, "pendingRate": 0.0, "platformFailureCount": 21, "successCount": 149, "successRate": 0.0743, - "total": 2005, - "unresolvedFailureCount": 1108 + "total": 2006, + "unresolvedFailureCount": 1109 }, "Iluvatar_mrv-100": { "attributableFailureCount": 964, @@ -4843,21 +4843,21 @@ "decisionSuccessRate": 0.0, "decisionTotal": 1, "failureBreakdown": { - "ambiguous_runtime": 156, + "ambiguous_runtime": 157, "memory_capacity": 1, "参数/模板问题": 2, "日志缺失": 1, "验证失败": 23 }, - "failureCount": 183, + "failureCount": 184, "failureRate": 1.0, "pendingCount": 0, "pendingRate": 0.0, "platformFailureCount": 0, "successCount": 0, "successRate": 0.0, - "total": 183, - "unresolvedFailureCount": 182 + "total": 184, + "unresolvedFailureCount": 183 }, "Kunlunxin_r-200-8f": { "attributableFailureCount": 0, @@ -7958,6 +7958,29 @@ "total": 1, "unresolvedFailureCount": 1 }, + "Iluvatar_bi-150|vllm_fix_tokenizer|text-generation|bailing_hybrid|none": { + "attributableFailureCount": 0, + "decisionFailureRate": 0.0, + "decisionSuccessRate": 0.0, + "decisionTotal": 0, + "failureBreakdown": { + "ambiguous_runtime": 1 + }, + "failureCount": 1, + "failureRate": 1.0, + "framework": "vllm_fix_tokenizer", + "modelType": "bailing_hybrid", + "pendingCount": 0, + "pendingRate": 0.0, + "platformFailureCount": 0, + "quantizationMethod": "none", + "successCount": 0, + "successRate": 0.0, + "targetGpu": "Iluvatar_bi-150", + "taskType": "text-generation", + "total": 1, + "unresolvedFailureCount": 1 + }, "Iluvatar_bi-150|vllm_fix_tokenizer|text-generation|gemma4_unified|none": { "attributableFailureCount": 0, "decisionFailureRate": 0.0, @@ -9208,10 +9231,10 @@ "decisionSuccessRate": 0.0, "decisionTotal": 0, "failureBreakdown": { - "ambiguous_runtime": 8, + "ambiguous_runtime": 9, "参数/模板问题": 1 }, - "failureCount": 9, + "failureCount": 10, "failureRate": 1.0, "framework": "vllm_fix_tokenizer", "modelType": "qwen3_5", @@ -9223,8 +9246,8 @@ "successRate": 0.0, "targetGpu": "Kunlunxin_p-800", "taskType": "text-generation", - "total": 9, - "unresolvedFailureCount": 9 + "total": 10, + "unresolvedFailureCount": 10 }, "Kunlunxin_p-800|vllm_fix_tokenizer|text-generation|qwen3|none": { "attributableFailureCount": 0, @@ -11370,7 +11393,7 @@ "failureRate": 1.0, "framework": "vllm_fix_tokenizer", "lastPlatformFailureAt": null, - "lastTerminalAt": "2026-09-20T13:30:42.155978+00:00", + "lastTerminalAt": "2026-09-20T14:47:45.260018+00:00", "pendingCount": 0, "pendingRate": 0.0, "platformFailureCount": 0, @@ -12476,6 +12499,31 @@ "total": 1, "unresolvedFailureCount": 1 }, + "Iluvatar_bi-150|vllm_fix_tokenizer|text-generation|bailing_hybrid|none": { + "attributableFailureCount": 0, + "consecutiveFailures": 0, + "decisionFailureRate": 0.0, + "decisionSuccessRate": 0.0, + "decisionTotal": 0, + "failureBreakdown": { + "ambiguous_runtime": 1 + }, + "failureCount": 1, + "failureRate": 1.0, + "framework": "vllm_fix_tokenizer", + "lastTerminalAt": "2026-09-20T14:47:45.260018+00:00", + "modelType": "bailing_hybrid", + "pendingCount": 0, + "pendingRate": 0.0, + "platformFailureCount": 0, + "quantizationMethod": "none", + "successCount": 0, + "successRate": 0.0, + "targetGpu": "Iluvatar_bi-150", + "taskType": "text-generation", + "total": 1, + "unresolvedFailureCount": 1 + }, "Iluvatar_bi-150|vllm_fix_tokenizer|text-generation|gemma4_unified|none": { "attributableFailureCount": 0, "consecutiveFailures": 0, @@ -17368,6 +17416,30 @@ "total": 1, "unresolvedFailureCount": 1 }, + "Iluvatar_bi-150|vllm_fix_tokenizer|text-generation|bailing_hybrid|none|33": { + "attributableFailureCount": 0, + "decisionFailureRate": 0.0, + "decisionSuccessRate": 0.0, + "decisionTotal": 0, + "failureBreakdown": { + "ambiguous_runtime": 1 + }, + "failureCount": 1, + "failureRate": 1.0, + "framework": "vllm_fix_tokenizer", + "loadSizeLog2Bucket": 33, + "modelType": "bailing_hybrid", + "pendingCount": 0, + "pendingRate": 0.0, + "platformFailureCount": 0, + "quantizationMethod": "none", + "successCount": 0, + "successRate": 0.0, + "targetGpu": "Iluvatar_bi-150", + "taskType": "text-generation", + "total": 1, + "unresolvedFailureCount": 1 + }, "Iluvatar_bi-150|vllm_fix_tokenizer|text-generation|gemma4_unified|none|33": { "attributableFailureCount": 0, "decisionFailureRate": 0.0, @@ -19244,9 +19316,9 @@ "decisionSuccessRate": 0.0, "decisionTotal": 0, "failureBreakdown": { - "ambiguous_runtime": 3 + "ambiguous_runtime": 4 }, - "failureCount": 3, + "failureCount": 4, "failureRate": 1.0, "framework": "vllm_fix_tokenizer", "loadSizeLog2Bucket": 33, @@ -19259,8 +19331,8 @@ "successRate": 0.0, "targetGpu": "Kunlunxin_p-800", "taskType": "text-generation", - "total": 3, - "unresolvedFailureCount": 3 + "total": 4, + "unresolvedFailureCount": 4 }, "Kunlunxin_p-800|vllm_fix_tokenizer|text-generation|qwen3_5|none|34": { "attributableFailureCount": 0, @@ -22195,15 +22267,15 @@ "unresolvedFailureCount": 0 } }, - "terminalRecords": 15870, - "totalRecords": 15975, + "terminalRecords": 15872, + "totalRecords": 15977, "totals": { "attributableFailureCount": 5755, "decisionFailureRate": 0.8614, "decisionSuccessRate": 0.1386, "decisionTotal": 6681, "failureBreakdown": { - "ambiguous_runtime": 3765, + "ambiguous_runtime": 3767, "architecture_compatibility": 212, "attention_backend": 1, "backend_operator": 102, @@ -22219,15 +22291,15 @@ "日志缺失": 719, "验证失败": 673 }, - "failureCount": 14944, + "failureCount": 14946, "failureRate": 0.9417, "pendingCount": 0, "pendingRate": 0.0, "platformFailureCount": 922, "successCount": 926, "successRate": 0.0583, - "total": 15870, - "unresolvedFailureCount": 8267 + "total": 15872, + "unresolvedFailureCount": 8269 }, "warnings": [ "GPU Iluvatar_bi-100 本地统计失败率偏高(≥50%),建议重点关注。", @@ -22272,12 +22344,12 @@ "组合 Cambricon_mlu-370-x4|unknown|visual-multi-modal 近期失败集中,建议降低该 GPU+框架的提交优先级。", "组合 Ascend_910-b4|vllm|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。", "组合 Iluvatar_bi-150|transformers|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。", - "组合 Cambricon_mlu-370-x8|vllm|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。", "组合 Cambricon_mlu-370-x8|unknown|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。", - "组合 Ascend_910-b3|vllm|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。" + "组合 Ascend_910-b3|vllm|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。", + "组合 Cambricon_mlu-370-x8|vllm|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。" ] }, "storageMode": "decision_state_only", - "summarizedRecords": 15975, + "summarizedRecords": 15977, "version": 1 } diff --git a/.modelhub_state/recent_outcomes.jsonl b/.modelhub_state/recent_outcomes.jsonl index bb579f47b..17037d3c4 100644 --- a/.modelhub_state/recent_outcomes.jsonl +++ b/.modelhub_state/recent_outcomes.jsonl @@ -7,6 +7,7 @@ {"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDetectedFramework": "vllm", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "framework": "vllm", "lastSyncTime": "2026-09-20T10:06:20.452855+00:00", "modelId": "KoboldAI/fairseq-dense-13B-Janeway", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T09:55:21+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4079951", "taskType": "text-generation", "verifyResult": -1} {"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "", "lastSyncTime": "2026-09-20T09:38:38.259110+00:00", "modelId": "logic65/Qwen3.8-Whittle-w50-18.3B-unrepaired", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T09:35:21+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4560061", "taskType": "text-generation", "verifyResult": -1} {"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["qwen3_5"], "framework": "", "lastSyncTime": "2026-09-20T06:27:54.872472+00:00", "modelId": "logic65/Qwen3.8-Whittle-w50-18.3B-unrepaired", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T06:21:21+00:00", "targetGpu": "Biren_166m", "taskId": "4610365", "taskType": "text-generation", "verifyResult": -1} +{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDetectedFramework": "vllm_fix_tokenizer", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-20T14:47:45.260018+00:00", "modelId": "inclusionAI/Ling-3.0-tiny", "modelProfile": {"architectures": ["BailingMoeV3ForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 15787992416, "estimatedRequiredGiB": 17.659, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "bailing_hybrid", "modelscopeFileSize": 15801179822, "modelscopeLicense": "mit", "modelscopeParams": 7893392800, "modelscopeTags": ["license:mit", "model_type:bailing_hybrid", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 15801179822}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T06:19:35.394755+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4972231", "taskType": "text-generation", "verifyResult": -1} {"failReason": null, "framework": "", "lastSyncTime": "2026-09-20T05:59:18.862303+00:00", "modelId": "AI-ModelScope/granite-20b-code-base-8k", "modelProfile": {}, "outcome": "success", "status": "success", "submitTime": "2026-09-20T05:57:22+00:00", "targetGpu": "MetaX_c-500", "taskId": "4079080", "taskType": "text-generation", "verifyResult": 1} {"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDetectedFramework": "transformers", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "framework": "transformers", "lastSyncTime": "2026-09-20T13:30:42.155956+00:00", "modelId": "Youssofal/Qwen3.5-9B-MTPLX-Optimized-Speed", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "3086a71adcdca90005b4967eb30749c44cd82389aff8d599758e4b91675d40dc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 8674988799, "estimatedRequiredGiB": 9.718, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 8695119291, "modelscopeLicense": "apache-2.0", "modelscopeParams": 2415484144, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:apple-silicon", "custom_tag:speculative-decoding", "custom_tag:qwen", "custom_tag:qwen3", "custom_tag:mtp", "custom_tag:mtplx", "custom_tag:local-ai", "custom_tag:qwen3-5", "custom_tag:9b", "custom_tag:6-bit"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 8695119291}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T05:16:13.358703+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4971314", "taskType": "text-generation", "verifyResult": -1} {"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDetectedFramework": "vllm", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "framework": "vllm", "lastSyncTime": "2026-09-20T13:30:42.156020+00:00", "modelId": "BAAI/RoboBrain2.5-8B-MT", "modelProfile": {"architectures": ["Qwen3VLForConditionalGeneration"], "configFingerprint": "ac4b2845ddd4bdce478dd2b65134ee3ab3d259a119afe6841bf068b135fd4892", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 17534339552, "estimatedRequiredGiB": 19.609, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_vl", "modelscopeFileSize": 17545918174, "modelscopeLicense": "apache-2.0", "modelscopeParams": 8767123696, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_vl", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 17545918174}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T05:16:13.351034+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4971313", "taskType": "text-generation", "verifyResult": -1} @@ -297,4 +298,3 @@ {"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["rwkv7"], "framework": "", "lastSyncTime": "2026-09-13T23:31:12.512853+00:00", "modelId": "RWKV/RWKV7-13.3B-20260805", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-13T23:25:21+00:00", "targetGpu": "Biren_166m", "taskId": "4610393", "taskType": "text-generation", "verifyResult": -1} {"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDetectedFramework": "vllm", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["rwkv7"], "framework": "vllm", "lastSyncTime": "2026-09-13T23:13:14.924714+00:00", "modelId": "RWKV/RWKV7-13.3B-20260805", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-13T23:09:21+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4592432", "taskType": "text-generation", "verifyResult": -1} {"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "", "lastSyncTime": "2026-09-13T21:45:41.613474+00:00", "modelId": "XHToken/Spark-X2.5-1.7B", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-13T21:39:21+00:00", "targetGpu": "Iluvatar_mrv-100", "taskId": "4559290", "taskType": "text-generation", "verifyResult": -1} -{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "", "lastSyncTime": "2026-09-13T21:36:16.709712+00:00", "modelId": "nm-testing/tinyllama-oneshot-w8a8-dynamic-token-v2-asym", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-13T21:33:21+00:00", "targetGpu": "Iluvatar_mrv-100", "taskId": "4523305", "taskType": "text-generation", "verifyResult": -1} diff --git a/.modelhub_state/recovery_intents.jsonl b/.modelhub_state/recovery_intents.jsonl index d3115c2f1..0e0c65cfd 100644 --- a/.modelhub_state/recovery_intents.jsonl +++ b/.modelhub_state/recovery_intents.jsonl @@ -1682,6 +1682,7 @@ {"batchId": "c208b2f645614a14a05f569686eaa8f0", "completedAt": "2026-09-20T14:43:55.731125+00:00", "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configSource": "modelhub_live", "createdAt": "2026-09-20T14:37:40.025509+00:00", "framework": "vllm_tokenizer_patch", "intentId": "cbea23846c1041d390683bf19aa2e1e3", "lastModified": "2026-08-24T19:06:55+00:00", "modelAddress": "https://modelscope.cn/models/RedHatAI/gemma-2-2b-it-quantized.w8a8", "reason": null, "reconciledAt": "2026-09-20T14:46:28.823206+00:00", "repoId": "RedHatAI/gemma-2-2b-it-quantized.w8a8", "safeConfigVector": {"gpuNum": 1}, "status": "recovered_active", "targetGpu": "Ascend_910-b4", "taskId": "4980558", "taskType": "text-generation"} {"batchId": "c208b2f645614a14a05f569686eaa8f0", "completedAt": "2026-09-20T14:43:55.731149+00:00", "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configSource": "modelhub_live", "createdAt": "2026-09-20T14:37:40.025941+00:00", "framework": "vllm_tokenizer_patch", "intentId": "945ed355fc854c3796bff67a188dc1a4", "lastModified": "2026-08-24T19:21:33+00:00", "modelAddress": "https://modelscope.cn/models/neuralmagic/starcoder2-3b-quantized.w8a8", "reason": null, "reconciledAt": "2026-09-20T14:46:28.824536+00:00", "repoId": "neuralmagic/starcoder2-3b-quantized.w8a8", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "recovered_active", "targetGpu": "Ascend_910-b3", "taskId": "4980575", "taskType": "text-generation"} {"batchId": "c208b2f645614a14a05f569686eaa8f0", "completedAt": "2026-09-20T14:43:55.731183+00:00", "configFingerprint": "1c0a4f69628b2a0ad37a6eee139a4c98afed7d8c93529a9ef42a90756e4e1940", "configSource": "modelhub_live", "createdAt": "2026-09-20T14:37:40.026474+00:00", "framework": "vllm_tokenizer_patch", "intentId": "cc61d16ed90546aa8cdaefcc5b85acf3", "lastModified": "2026-08-24T19:39:21+00:00", "modelAddress": "https://modelscope.cn/models/neuralmagic/gemma-2-9b-it-quantized.w8a16", "reason": null, "reconciledAt": "2026-09-20T14:46:28.824533+00:00", "repoId": "neuralmagic/gemma-2-9b-it-quantized.w8a16", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "recovered_active", "targetGpu": "Kunlunxin_p-800", "taskId": "4980576", "taskType": "text-generation"} +{"batchId": "feb78ef0c18e4e3985359b64b1296b80", "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configSource": "modelhub_live", "createdAt": "2026-09-20T14:48:03.991748+00:00", "framework": "vllm-customized", "intentId": "71511ea3fb1c4043b4770c7db4bb7a60", "lastModified": "2026-09-14T13:40:21+00:00", "modelAddress": "https://modelscope.cn/models/IntervitensInc/kek_mk3", "repoId": "IntervitensInc/kek_mk3", "safeConfigVector": {"dtype": "-", "gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Cambricon_mlu-370-x8", "taskType": "text-generation"} {"batchId": "25ba0a4da7e34622a6702513d8cdc4ce", "completedAt": "2026-09-20T14:46:26.841216+00:00", "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configSource": "modelhub_live", "createdAt": "2026-09-20T14:43:59.030327+00:00", "framework": "vllm", "intentId": "6552e2f86a624809a049273684b0db69", "repoId": "aisingapore/Qwen-SEA-LION-v4-32B-IT-4BIT", "safeConfigVector": {"dtype": "-", "gpuNum": 1, "maxModelLen": 2048}, "status": "age_policy_deferred", "targetGpu": "Ascend_910-b4", "taskId": null, "taskType": "text-generation"} {"batchId": "25ba0a4da7e34622a6702513d8cdc4ce", "completedAt": "2026-09-20T14:46:26.841213+00:00", "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configSource": "modelhub_live", "createdAt": "2026-09-20T14:43:59.030267+00:00", "framework": "vllm", "intentId": "d4a6ec4d041443a3aa7cbe85d6570cfb", "repoId": "RedHatAI/gemma-2-2b-it-quantized.w8a16", "safeConfigVector": {"dtype": "-", "gpuNum": 1, "maxModelLen": 2048}, "status": "age_policy_deferred", "targetGpu": "Ascend_910-b4", "taskId": null, "taskType": "text-generation"} {"batchId": "25ba0a4da7e34622a6702513d8cdc4ce", "completedAt": "2026-09-20T14:46:26.841210+00:00", "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configSource": "modelhub_live", "createdAt": "2026-09-20T14:43:59.030208+00:00", "framework": "vllm", "intentId": "c4efb924449747e090a8b2f171754135", "repoId": "RedHatAI/starcoder2-15b-FP8", "safeConfigVector": {"dtype": "-", "gpuNum": 1, "maxModelLen": 2048}, "status": "age_policy_deferred", "targetGpu": "Ascend_910-b4", "taskId": null, "taskType": "text-generation"} diff --git a/manifest.json b/manifest.json index 35ceb46fd..5b3474c91 100644 --- a/manifest.json +++ b/manifest.json @@ -1,24 +1,24 @@ { "agentVersion": "2026.09.20.2", "checksums": { - ".modelhub_state/architecture_compatibility_blacklist.json": "8ce8e70aa6161c0a647d0d3125ff96f646bd28261449c828f013ecad7f94def0", + ".modelhub_state/architecture_compatibility_blacklist.json": "031ff79178960a15f140270afad755c1a0249cd31f7c7938821e81c51108f5eb", ".modelhub_state/architecture_history_backfill.json": "a92348206605da04f75c9e26e7c818b76f259e0b28983f18b6375ef94802f0ab", - ".modelhub_state/market_intelligence.json": "d50e24afb465ba0a53a458afeeb3bd7b24c8a6d5b57beba47a38e67ad48100f0", - ".modelhub_state/official_capabilities.json": "c1a7135409969cbfb8497079a132c418f63500ff634f656e9205f69d3cbecd73", - ".modelhub_state/outcome_checkpoint.json": "a667cedf73e9a0f4f2814e50453c7573a5c6348ac3774c2e317e970dea261358", + ".modelhub_state/market_intelligence.json": "32b3405fc94107c7232b8ee45e943e3dcd07652617fe1d08452c59ab1671cf65", + ".modelhub_state/official_capabilities.json": "75d2529478caba51de5aa72e3c058825fa930cc23dfe4fc8f9b489f33395290b", + ".modelhub_state/outcome_checkpoint.json": "6d7782c8c9fc862f97956cde907b6d480a0f7a456ac794050e320f9ed7087a41", ".modelhub_state/queue_cleanup_latest.json": "eb01f106d0cd4018a0346c4a81182d6a9e67f5d4d14db22f5d0c685667977bc3", - ".modelhub_state/recent_outcomes.jsonl": "87a4b35ffc2fc9627481aac3fd30838519ab63cabfcc42af6e4e9f245273b425", + ".modelhub_state/recent_outcomes.jsonl": "c686c563ab48519884fd42d9952589ca0656ee5a5fab0f1556f4f5dc80e978ff", ".modelhub_state/recovery_active_tasks.jsonl": "9e0d5d0e0ca2e6fe45bf814462091e89c6625c7005e5953f84cfe1ed47a7d8fc", - ".modelhub_state/recovery_intents.jsonl": "eb7c7b303725020867f706fad177454ba4c2b34891fc1c37266a5b4165cc8e35", + ".modelhub_state/recovery_intents.jsonl": "aafa001d637156e6d3e306648d301fc077a04c4aa74f131f6a7dc9d258671531", ".modelhub_state/routing_intelligence.json": "87ca00c8bdf8c0f050aa5b51768054f887caeeb2cc3d9144da6a05b86ddde4e3", ".modelhub_state/submission_exclusions.jsonl": "221be895a524330815b3c5124450b33c94b5f1e68169030c73684bf675d06ec7", ".modelhub_state/worker_crashes.jsonl": "a494412048eca6dce83e7284303a6b4b56a445c1274ac201ab8723c739a14f23", "ledger/submissions.jsonl": "f5217e09fdbe3d527514101f8ffc3469765469f1e06305c1a32a40055965d60c", - "outcomes/submissions.jsonl": "38daa33f6497e7bcfde1cae8a68cf10cf0d7e6283cc65d5a7b9b31002845238e" + "outcomes/submissions.jsonl": "9dd091c37ef249704e36df327ea233cc74c00589fc1e72b8a5e8e170e8af36c8" }, - "generation": 10634, - "phase": "cycle", + "generation": 10635, + "phase": "intent", "schemaVersion": 1, - "updatedAt": "2026-09-20T14:46:44.607778+00:00", + "updatedAt": "2026-09-20T14:48:04.134747+00:00", "writerId": "fc715c06e24c4db2ae3e425e435db996" } diff --git a/outcomes/submissions.jsonl b/outcomes/submissions.jsonl index c98a0a5fb..f36f19034 100644 --- a/outcomes/submissions.jsonl +++ b/outcomes/submissions.jsonl @@ -158,7 +158,6 @@ {"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-10T14:29:51.626244+00:00", "modelId": "CohereLabs/tiny-aya-en-thinker", "modelProfile": {"architectures": ["Cohere2ForCausalLM"], "configFingerprint": "71d6e16091ba0289f01acffc8e39bb8c25703488dc49b9e390e08951fe1f2b33", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 6700539432, "estimatedRequiredGiB": 7.522, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "cohere2", "modelscopeFileSize": 6730844445, "modelscopeLicense": "cc-by-nc-4.0", "modelscopeParams": 3350243328, "modelscopeTags": ["license:cc-by-nc-4.0", "model_type:cohere2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:aya", "custom_tag:multilingual", "custom_tag:reasoning", "custom_tag:tiny-aya"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 6730844445}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-10T06:28:56.134725+00:00", "targetGpu": "Iluvatar_mrv-100", "taskId": "4754659", "taskType": "text-generation", "verifyResult": null} {"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-10T15:03:19.195695+00:00", "modelId": "CohereLabs/tiny-aya-en-thinker", "modelProfile": {"architectures": ["Cohere2ForCausalLM"], "configFingerprint": "cbe5569cdb6c42553436b837dd5b8f0cfa03334642a56f4ff0137f952f8eb84b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 6700539432, "estimatedRequiredGiB": 7.522, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "cohere2", "modelscopeFileSize": 6730844445, "modelscopeLicense": "cc-by-nc-4.0", "modelscopeParams": 3350243328, "modelscopeTags": ["license:cc-by-nc-4.0", "model_type:cohere2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:aya", "custom_tag:multilingual", "custom_tag:reasoning", "custom_tag:tiny-aya"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 6730844445}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-10T06:56:12.042564+00:00", "targetGpu": "Vastai_va16", "taskId": "4755190", "taskType": "text-generation", "verifyResult": null} {"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-10T15:19:40.009053+00:00", "modelId": "primitive-ai/Nex-N2.5-mini-FP8", "modelProfile": {"architectures": ["Qwen3_5MoeForConditionalGeneration"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 38136526544, "estimatedRequiredGiB": 42.65, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_moe", "modelscopeFileSize": 38162746086, "modelscopeLicense": "apache-2.0", "modelscopeParams": 35107181936, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5_moe", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:vllm", "custom_tag:compressed-tensors", "custom_tag:fp8", "custom_tag:quantized", "custom_tag:moe", "custom_tag:nex-n2.5", "custom_tag:agentic", "custom_tag:tool-calling", "custom_tag:single-gpu", "custom_tag:blackwell"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 38162746086}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-10T07:12:26.204766+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4755596", "taskType": "text-generation", "verifyResult": null} -{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-10T15:19:40.009069+00:00", "modelId": "BAAI/AREX-Turbo", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 9078620368, "estimatedRequiredGiB": 10.18, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 9109259594, "modelscopeLicense": "apache-2.0", "modelscopeParams": 4539265536, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:agent", "custom_tag:deep-research", "custom_tag:reasoning", "custom_tag:tool-use", "custom_tag:long-context", "custom_tag:qwen3.5", "custom_tag:dense"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "recentProfileFeedback": {"attributableFailureCount": 0, "consecutiveFailures": 0, "decisionFailureRate": 0.0, "decisionSuccessRate": 0.0, "decisionTotal": 0, "failureBreakdown": {"ambiguous_runtime": 1}, "failureCount": 1, "failureRate": 1.0, "framework": "vllm_fix_tokenizer", "lastTerminalAt": "2026-09-09T14:21:21.949441+00:00", "modelType": "qwen3_5", "pendingCount": 0, "pendingRate": 0.0, "platformFailureCount": 0, "quantizationMethod": "none", "successCount": 0, "successRate": 0.0, "targetGpu": "Kunlunxin_p-800", "taskType": "text-generation", "total": 1, "unresolvedFailureCount": 1}, "repositoryOnDiskBytes": 9109259594}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-10T07:12:26.214438+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4755598", "taskType": "text-generation", "verifyResult": null} {"failReason": null, "framework": "vllm-patch-tokenizer", "lastSyncTime": "2026-09-10T15:40:28.508808+00:00", "modelId": "CohereLabs/tiny-aya-en-thinker", "modelProfile": {"architectures": ["Cohere2ForCausalLM"], "configFingerprint": "0d5bc82787a89d4313672808d1a8629a1fc675c91aa6a08865f770aa24447fbc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 6700539432, "estimatedRequiredGiB": 7.522, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "cohere2", "modelscopeFileSize": 6730844445, "modelscopeLicense": "cc-by-nc-4.0", "modelscopeParams": 3350243328, "modelscopeTags": ["license:cc-by-nc-4.0", "model_type:cohere2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:aya", "custom_tag:multilingual", "custom_tag:reasoning", "custom_tag:tiny-aya"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 6730844445}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-10T07:33:41.787375+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4756078", "taskType": "text-generation", "verifyResult": null} {"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-10T15:56:47.546297+00:00", "modelId": "CohereLabs/tiny-aya-en-thinker", "modelProfile": {"architectures": ["Cohere2ForCausalLM"], "configFingerprint": "a494b1667fab53609b94cc853db1a12c8d3b9fcf98795ecbab3dd5eb3a3c77f6", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 6700539432, "estimatedRequiredGiB": 7.522, "gpuMemoryEvidence": {"memoryGiB": 48.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 48.0, "maximumRepositorySizeGiB": 40.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "cohere2", "modelscopeFileSize": 6730844445, "modelscopeLicense": "cc-by-nc-4.0", "modelscopeParams": 3350243328, "modelscopeTags": ["license:cc-by-nc-4.0", "model_type:cohere2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:aya", "custom_tag:multilingual", "custom_tag:reasoning", "custom_tag:tiny-aya"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 6730844445}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-10T07:50:25.385741+00:00", "targetGpu": "Mthreads_s4000", "taskId": "4756398", "taskType": "text-generation", "verifyResult": null} {"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-10T21:33:23.930689+00:00", "modelId": "primitive-ai/Nex-N2.5-mini-FP8", "modelProfile": {"architectures": ["Qwen3_5MoeForConditionalGeneration"], "configFingerprint": "d341b8002cf39d3299fd429479e8c8e954583e49d4fec5a640ddc0f8f0cbd054", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 38136526544, "estimatedRequiredGiB": 42.65, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_moe", "modelscopeFileSize": 38162746086, "modelscopeLicense": "apache-2.0", "modelscopeParams": 35107181936, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5_moe", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:vllm", "custom_tag:compressed-tensors", "custom_tag:fp8", "custom_tag:quantized", "custom_tag:moe", "custom_tag:nex-n2.5", "custom_tag:agentic", "custom_tag:tool-calling", "custom_tag:single-gpu", "custom_tag:blackwell"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 38162746086}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-10T13:27:23.406185+00:00", "targetGpu": "Biren_166m", "taskId": "4760702", "taskType": "text-generation", "verifyResult": null} @@ -975,11 +974,10 @@ {"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T14:18:03.259019+00:00", "modelId": "RedHatAI/gemma-2-2b-quantized.w8a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 6746735136, "estimatedRequiredGiB": 7.565, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 6768615358, "modelscopeLicense": "gemma", "modelscopeParams": 3204165888, "modelscopeTags": ["license:gemma", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 6768615358}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T06:16:07.298594+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4972201", "taskType": "text-generation", "verifyResult": null} {"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-20T14:18:03.258951+00:00", "modelId": "RedHatAI/gemma-2-27b-it-quantized.w8a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 30775446656, "estimatedRequiredGiB": 34.419, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 30797371689, "modelscopeLicense": "gemma", "modelscopeParams": 28406776320, "modelscopeTags": ["license:gemma", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 30797371689}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T06:16:07.269550+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4972172", "taskType": "text-generation", "verifyResult": null} {"failReason": null, "framework": "vllm", "lastSyncTime": null, "modelId": "neuralmagic/Qwen2-0.5B-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["Qwen2ForCausalLM"], "configFingerprint": "d341b8002cf39d3299fd429479e8c8e954583e49d4fec5a640ddc0f8f0cbd054", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 903168128, "estimatedRequiredGiB": 1.022, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen2", "modelscopeFileSize": 914658310, "modelscopeLicense": "apache-2.0", "modelscopeParams": 630167424, "modelscopeTags": ["license:apache-2.0", "model_type:qwen2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 914658310}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-20T06:17:57.834294+00:00", "targetGpu": "Biren_166m", "taskId": "4972230", "taskType": "text-generation", "verifyResult": null} -{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": null, "modelId": "inclusionAI/Ling-3.0-tiny", "modelProfile": {"architectures": ["BailingMoeV3ForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 15787992416, "estimatedRequiredGiB": 17.659, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "bailing_hybrid", "modelscopeFileSize": 15801179822, "modelscopeLicense": "mit", "modelscopeParams": 7893392800, "modelscopeTags": ["license:mit", "model_type:bailing_hybrid", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 15801179822}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-20T06:19:35.394755+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4972231", "taskType": "text-generation", "verifyResult": null} {"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T14:32:15.260743+00:00", "modelId": "neuralmagic/Llama-3.2-1B-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2024670536, "estimatedRequiredGiB": 2.273, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 2033824885, "modelscopeLicense": "llama3.2", "modelscopeParams": 1498482688, "modelscopeTags": ["license:llama3.2", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:llama", "custom_tag:llama-3", "custom_tag:neuralmagic", "custom_tag:llmcompressor"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 2033824885}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T06:31:52.552038+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4972399", "taskType": "text-generation", "verifyResult": null} {"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T14:34:57.856700+00:00", "modelId": "neuralmagic/gemma-2-9b-it-quantized.w4a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7963207224, "estimatedRequiredGiB": 8.924, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 7985124714, "modelscopeLicense": "llama2", "modelscopeParams": 10159209984, "modelscopeTags": ["license:llama2", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 7985124714}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T06:32:07.434077+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4972417", "taskType": "text-generation", "verifyResult": null} {"failReason": null, "framework": "vllm-mlu", "lastSyncTime": "2026-09-20T14:34:57.856737+00:00", "modelId": "RedHatAI/Phi-3-medium-128k-instruct-FP8", "modelProfile": {"architectures": ["Phi3ForCausalLM"], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 14289052464, "estimatedRequiredGiB": 15.972, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "phi3", "modelscopeFileSize": 14291553638, "modelscopeLicense": "mit", "modelscopeParams": 13960238080, "modelscopeTags": ["license:mit", "model_type:phi3", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:fp8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 14291553638}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T06:33:31.623471+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4972418", "taskType": "text-generation", "verifyResult": null} -{"failReason": null, "framework": "vllm", "lastSyncTime": null, "modelId": "neuralmagic/Mistral-Nemo-Instruct-2407-quantized.w4a16", "modelProfile": {"architectures": ["MistralForCausalLM"], "configFingerprint": "6db01b5037d5ed2db74875d2113f8582c97f34bd825637161e54174c9f2cabed", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 8308282608, "estimatedRequiredGiB": 9.319, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "mistral", "modelscopeFileSize": 8338221588, "modelscopeLicense": "llama2", "modelscopeParams": 12247782400, "modelscopeTags": ["license:llama2", "model_type:mistral", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 8338221588}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-20T06:34:00.535697+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4972450", "taskType": "text-generation", "verifyResult": null} +{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T14:47:45.260008+00:00", "modelId": "neuralmagic/Mistral-Nemo-Instruct-2407-quantized.w4a16", "modelProfile": {"architectures": ["MistralForCausalLM"], "configFingerprint": "6db01b5037d5ed2db74875d2113f8582c97f34bd825637161e54174c9f2cabed", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 8308282608, "estimatedRequiredGiB": 9.319, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "mistral", "modelscopeFileSize": 8338221588, "modelscopeLicense": "llama2", "modelscopeParams": 12247782400, "modelscopeTags": ["license:llama2", "model_type:mistral", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 8338221588}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T06:34:00.535697+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4972450", "taskType": "text-generation", "verifyResult": null} {"failReason": null, "framework": "vllm-mlu", "lastSyncTime": null, "modelId": "neuralmagic/Llama-3.2-1B-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "36ec94e289a8da5fd70b98a2f401fc5159740daf9e2a076da4fd490f34bcbc0b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2024670536, "estimatedRequiredGiB": 2.273, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 2033824885, "modelscopeLicense": "llama3.2", "modelscopeParams": 1498482688, "modelscopeTags": ["license:llama3.2", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:llama", "custom_tag:llama-3", "custom_tag:neuralmagic", "custom_tag:llmcompressor"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 2033824885}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-20T06:49:15.390546+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "4972568", "taskType": "text-generation", "verifyResult": null} {"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": null, "modelId": "neuralmagic/Meta-Llama-3.1-8B-quantized.w8a8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 9084013360, "estimatedRequiredGiB": 10.162, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 9093204103, "modelscopeLicense": "llama3.1", "modelscopeParams": 8030261248, "modelscopeTags": ["license:llama3.1", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm", "custom_tag:quantized", "custom_tag:8-bit"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 9093204103}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-20T07:19:46.762739+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4972960", "taskType": "text-generation", "verifyResult": null} {"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": null, "modelId": "amd/gpt-oss-20b-BF16-w8a8-llmcompressor", "modelProfile": {"architectures": ["GptOssForCausalLM"], "configFingerprint": "1c0a4f69628b2a0ad37a6eee139a4c98afed7d8c93529a9ef42a90756e4e1940", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 44193251544, "estimatedRequiredGiB": 49.422, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gpt_oss", "modelscopeFileSize": 44221632439, "modelscopeLicense": "apache-2.0", "modelscopeParams": 20914757184, "modelscopeTags": ["license:apache-2.0", "model_type:gpt_oss", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:quantized", "custom_tag:int8", "custom_tag:w8a8", "custom_tag:dynamic-quantization", "custom_tag:8-bit", "custom_tag:llm-compressor", "custom_tag:compressed-tensors", "custom_tag:zendnn", "custom_tag:amd", "custom_tag:cpu-inference"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 44221632439}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-20T07:19:46.756927+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4972959", "taskType": "text-generation", "verifyResult": null}