state: generation 11665 (intent)
This commit is contained in:
@@ -17,6 +17,7 @@
|
||||
{"failReason": "参数/模板问题", "framework": "", "lastSyncTime": "2026-09-21T06:41:38.664002+00:00", "modelId": "ddalcu/Qwen3.8-27B-MLX-Serve-iQ-MLX-3.8bpw", "modelProfile": {}, "outcome": "failed", "status": "failed", "submitTime": "2026-09-21T06:27:41+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4783202", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": "参数/模板问题", "framework": "", "lastSyncTime": "2026-09-21T06:41:38.663896+00:00", "modelId": "ornith-ai/Ornith-1.5-9B-MLX-6bit", "modelProfile": {}, "outcome": "failed", "status": "failed", "submitTime": "2026-09-21T06:27:40+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4920239", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": "参数/模板问题", "framework": "", "lastSyncTime": "2026-09-21T06:41:38.663932+00:00", "modelId": "rapid-mlx/Qwen3.8-27B-4bit-MTP-MLX", "modelProfile": {}, "outcome": "failed", "status": "failed", "submitTime": "2026-09-21T06:27:40+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4920434", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "driver_error", "failureCode": "DRIVER_ERROR", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-21T19:49:37.657119+00:00", "modelId": "nm-testing/Meta-Llama-3-8B-Instruct-W8A8-Dyn-Per-Token", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 9084013064, "estimatedRequiredGiB": 10.162, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 9093197403, "modelscopeLicense": null, "modelscopeParams": 8030261248, "modelscopeTags": ["model_type:llama", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "recentProfileFeedback": {"attributableFailureCount": 0, "consecutiveFailures": 0, "decisionFailureRate": 0.0, "decisionSuccessRate": 0.0, "decisionTotal": 0, "failureBreakdown": {"ambiguous_runtime": 1}, "failureCount": 1, "failureRate": 1.0, "framework": "vllm_tokenizer_patch", "lastTerminalAt": "2026-09-20T18:18:38.666111+00:00", "modelType": "llama", "pendingCount": 0, "pendingRate": 0.0, "platformFailureCount": 0, "quantizationMethod": "compressed-tensors", "successCount": 0, "successRate": 0.0, "targetGpu": "Ascend_910-b4", "taskType": "text-generation", "total": 1, "unresolvedFailureCount": 1}, "repositoryOnDiskBytes": 9093197403}, "outcome": "failed", "status": "success", "submitTime": "2026-09-21T06:25:35.859746+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4995261", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "参数/模板问题", "framework": "", "lastSyncTime": "2026-09-21T05:59:02.567443+00:00", "modelId": "RedHatAI/gemma-4-12B-it-FP8-Dynamic", "modelProfile": {}, "outcome": "failed", "status": "failed", "submitTime": "2026-09-21T05:55:34+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4968995", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "", "lastSyncTime": "2026-09-21T05:51:50.448464+00:00", "modelId": "BAAI/Hospitality-llama3_1_8B_instruct", "modelProfile": {}, "outcome": "success", "status": "success", "submitTime": "2026-09-21T05:49:22+00:00", "targetGpu": "Sunrise_pt-200-x1", "taskId": "4457981", "taskType": "text-generation", "verifyResult": 1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-21T05:42:03.443622+00:00", "modelId": "tencent-community/WeDLM-7B-Instruct", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-21T05:29:21+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4079933", "taskType": "text-generation", "verifyResult": -1}
|
||||
@@ -63,6 +64,7 @@
|
||||
{"failReason": null, "framework": "", "lastSyncTime": "2026-09-21T02:49:19.765032+00:00", "modelId": "BAAI/Aerospace-llama3_1_8B_instruct", "modelProfile": {}, "outcome": "success", "status": "success", "submitTime": "2026-09-21T02:47:22+00:00", "targetGpu": "Sunrise_pt-200-x1", "taskId": "4457982", "taskType": "text-generation", "verifyResult": 1}
|
||||
{"failReason": "runtime_memory", "failureAction": "lower_memory_risk_or_reject_combination", "failureCategory": "runtime_memory", "failureClassificationReason": "structured_device_oom", "failureCode": "DEVICE_OOM", "failureDetectedFramework": "vllm", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu", "framework": "vllm", "lastSyncTime": "2026-09-21T02:38:57.372787+00:00", "modelId": "nv-community/Llama-3_3-Nemotron-Super-49B-v1_5-NVFP4", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-21T02:37:21+00:00", "targetGpu": "MetaX_c-500", "taskId": "4079201", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": null, "framework": "", "lastSyncTime": "2026-09-21T02:15:17.867390+00:00", "modelId": "BAAI/Automobile-llama3_1_8B_instruct", "modelProfile": {}, "outcome": "success", "status": "success", "submitTime": "2026-09-21T02:07:21+00:00", "targetGpu": "Sunrise_pt-200-x1", "taskId": "4457986", "taskType": "text-generation", "verifyResult": 1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm-customized", "lastSyncTime": "2026-09-21T19:49:37.657075+00:00", "modelId": "mlx-community/LFM2.5-1.2B-Instruct-4bit", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 658540250, "estimatedRequiredGiB": 0.741, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 663396475, "modelscopeLicense": "other", "modelscopeParams": 182975232, "modelscopeTags": ["license:other", "model_type:lfm2", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge", "custom_tag:mlx"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 663396475}, "outcome": "failed", "status": "success", "submitTime": "2026-09-21T01:45:52.835844+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4991428", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "model_load", "failureAction": "inspect_root_exception_then_llm_if_unknown", "failureCategory": "model_load", "failureClassificationReason": "broad_load_error", "failureCode": "MODEL_LOAD_FAILED", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "model_gpu_framework", "framework": "vllm", "lastSyncTime": "2026-09-21T01:35:45.556193+00:00", "modelId": "vllm-ascend/Qwen3-32B-w8a8sc-310-vllm-tp4", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-21T01:23:21+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4079972", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T18:20:35.656980+00:00", "modelId": "neuralmagic/starcoder2-15b-quantized.w8a8", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 17785240496, "estimatedRequiredGiB": 19.88, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 17788612744, "modelscopeLicense": "bigcode-openrail-m", "modelscopeParams": 15957889024, "modelscopeTags": ["license:bigcode-openrail-m", "model_type:starcoder2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 17788612744}, "outcome": "failed", "status": "success", "submitTime": "2026-09-21T00:12:58.737197+00:00", "targetGpu": "Biren_166m", "taskId": "4990141", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-21T15:54:37.859021+00:00", "modelId": "neuralmagic/Meta-Llama-3.1-8B-quantized.w8a16", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "d341b8002cf39d3299fd429479e8c8e954583e49d4fec5a640ddc0f8f0cbd054", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 9084040952, "estimatedRequiredGiB": 10.163, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 9093251901, "modelscopeLicense": "llama3.1", "modelscopeParams": 8030261248, "modelscopeTags": ["license:llama3.1", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 9093251901}, "outcome": "failed", "status": "success", "submitTime": "2026-09-21T00:00:11.334848+00:00", "targetGpu": "Biren_166m", "taskId": "4989991", "taskType": "text-generation", "verifyResult": -1}
|
||||
@@ -74,6 +76,7 @@
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T16:50:30.559291+00:00", "modelId": "TokenRhythm/NeoHorse-1-4B", "modelProfile": {"architectures": ["Qwen3_5ForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 8411552560, "estimatedRequiredGiB": 9.428, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_text", "modelscopeFileSize": 8435794426, "modelscopeLicense": "Apache License 2.0", "modelscopeParams": 4205751296, "modelscopeTags": ["license:Apache License 2.0", "model_type:qwen3_5_text", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:agentic", "custom_tag:tool-use", "custom_tag:coding", "custom_tag:reasoning", "custom_tag:instruction-following"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 8435794426}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T22:29:12.193862+00:00", "targetGpu": "Biren_166m", "taskId": "4988754", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T19:05:56.949036+00:00", "modelId": "sbintuitions/sarashina2.2-3b-instruct-v0.1", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 6711252896, "estimatedRequiredGiB": 7.503, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 6713123803, "modelscopeLicense": "mit", "modelscopeParams": 3355609600, "modelscopeTags": ["license:mit", "model_type:llama", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 6713123803}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T22:29:12.187825+00:00", "targetGpu": "Biren_166m", "taskId": "4988752", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "runtime_memory", "failureAction": "lower_memory_risk_or_reject_combination", "failureCategory": "runtime_memory", "failureClassificationReason": "structured_device_oom", "failureCode": "DEVICE_OOM", "failureDetectedFramework": "vllm", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu", "framework": "vllm", "lastSyncTime": "2026-09-20T22:20:31.465723+00:00", "modelId": "siliconflow/Qwen3-Coder-30B-A3B-Instruct-fp8", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T22:19:21+00:00", "targetGpu": "Mthreads_s4000", "taskId": "4587221", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm-customized", "lastSyncTime": "2026-09-21T19:49:37.657146+00:00", "modelId": "RedHatAI/Meta-Llama-3.1-8B-quantized.w8a16", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 9084040952, "estimatedRequiredGiB": 10.163, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 9093251763, "modelscopeLicense": "llama3.1", "modelscopeParams": 8030261248, "modelscopeTags": ["license:llama3.1", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "recentProfileFeedback": {"attributableFailureCount": 0, "consecutiveFailures": 0, "decisionFailureRate": 0.0, "decisionSuccessRate": 0.0, "decisionTotal": 0, "failureBreakdown": {"ambiguous_runtime": 1}, "failureCount": 1, "failureRate": 1.0, "framework": "vllm-customized", "lastTerminalAt": "2026-09-20T20:49:19.683358+00:00", "modelType": "llama", "pendingCount": 0, "pendingRate": 0.0, "platformFailureCount": 0, "quantizationMethod": "compressed-tensors", "successCount": 0, "successRate": 0.0, "targetGpu": "Cambricon_mlu-370-x8", "taskType": "text-generation", "total": 1, "unresolvedFailureCount": 1}, "repositoryOnDiskBytes": 9093251763}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T21:24:36.675557+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4987943", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDetectedFramework": "vllm_fix_tokenizer", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T05:42:03.443541+00:00", "modelId": "RedHatAI/Phi-3-medium-128k-instruct-FP8", "modelProfile": {"architectures": ["Phi3ForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 14289052464, "estimatedRequiredGiB": 15.972, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "phi3", "modelscopeFileSize": 14291553638, "modelscopeLicense": "mit", "modelscopeParams": 13960238080, "modelscopeTags": ["license:mit", "model_type:phi3", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:fp8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 14291553638}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T21:19:40.047378+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4987847", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T19:05:56.949080+00:00", "modelId": "LiquidAI/LFM2.5-1.2B-Instruct-MLX-bf16", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2340697867, "estimatedRequiredGiB": 2.622, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 2345693224, "modelscopeLicense": "other", "modelscopeParams": 1170340608, "modelscopeTags": ["license:other", "model_type:lfm2", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge", "custom_tag:mlx"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 2345693224}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T21:19:10.977747+00:00", "targetGpu": "Biren_166m", "taskId": "4987874", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T19:05:56.949051+00:00", "modelId": "primitive-ai/Qwen3.8-27B-mixed-NVFP4-FP8", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 22210552448, "estimatedRequiredGiB": 24.849, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 22234091413, "modelscopeLicense": "apache-2.0", "modelscopeParams": 17463440388, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:vllm", "custom_tag:compressed-tensors", "custom_tag:nvfp4", "custom_tag:fp8", "custom_tag:mixed-precision", "custom_tag:quantized", "custom_tag:speculative-decoding", "custom_tag:blackwell", "custom_tag:a100"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 22234091413}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T21:15:43.553144+00:00", "targetGpu": "Biren_166m", "taskId": "4987832", "taskType": "text-generation", "verifyResult": -1}
|
||||
@@ -161,6 +164,7 @@
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-21T01:08:25.664308+00:00", "modelId": "RedHatAI/Meta-Llama-3.1-8B-quantized.w8a16", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "d341b8002cf39d3299fd429479e8c8e954583e49d4fec5a640ddc0f8f0cbd054", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 9084040952, "estimatedRequiredGiB": 10.163, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 9093251763, "modelscopeLicense": "llama3.1", "modelscopeParams": 8030261248, "modelscopeTags": ["license:llama3.1", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 9093251763}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T11:54:11.089175+00:00", "targetGpu": "Biren_166m", "taskId": "4978283", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "tokenizer_compatibility", "failureAction": "check_tokenizer_files_then_llm", "failureCategory": "tokenizer_compatibility", "failureClassificationReason": "broad_tokenizer_error", "failureCode": "TOKENIZER_FAILED", "failureDetectedFramework": "transformers", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "model_framework", "framework": "transformers", "lastSyncTime": "2026-09-20T19:56:29.962428+00:00", "modelId": "mlx-community/Qwen3.5-35B-A3B-OptiQ-4bit-REAP-19B", "modelProfile": {"architectures": ["Qwen3_5MoeForConditionalGeneration"], "configFingerprint": "3086a71adcdca90005b4967eb30749c44cd82389aff8d599758e4b91675d40dc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 13736540899, "estimatedRequiredGiB": 15.375, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_moe", "modelscopeFileSize": 13756990943, "modelscopeLicense": "apache-2.0", "modelscopeParams": 3825799024, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5_moe", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:quantized", "custom_tag:expert-pruning", "custom_tag:reap", "custom_tag:moe", "custom_tag:optiq", "custom_tag:apple-silicon", "custom_tag:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 13756990943}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T11:44:01.235722+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4978143", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDetectedFramework": "vllm", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["qwen3_5"], "framework": "vllm", "lastSyncTime": "2026-09-20T11:30:54.600461+00:00", "modelId": "logic65/Qwen3.8-Whittle-w50-18.3B-unrepaired", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T11:23:21+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4523512", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-21T19:49:37.657133+00:00", "modelId": "neuralmagic/SmolLM-1.7B-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "6db01b5037d5ed2db74875d2113f8582c97f34bd825637161e54174c9f2cabed", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2014789064, "estimatedRequiredGiB": 2.255, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 2018175215, "modelscopeLicense": "apache-2.0", "modelscopeParams": 1812039680, "modelscopeTags": ["license:apache-2.0", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 2018175215}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T10:51:35.807777+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4977519", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "missing_structured_error_code", "failureCode": null, "failureDetectedFramework": "transformers", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "transformers", "lastSyncTime": "2026-09-20T21:06:45.850234+00:00", "modelId": "XHToken/Spark-X2.5-1.7B", "modelProfile": {"architectures": ["Spark2_5ForCausalLM"], "configFingerprint": "3086a71adcdca90005b4967eb30749c44cd82389aff8d599758e4b91675d40dc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 3415340496, "estimatedRequiredGiB": 3.834, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "spark2_5", "modelscopeFileSize": 3430847031, "modelscopeLicense": "apache-2.0", "modelscopeParams": 1707657216, "modelscopeTags": ["license:apache-2.0", "model_type:spark2_5", "library:safetensors", "library:", "task:text-generation", "custom_tag:llm", "custom_tag:sparkx2_5"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 3430847031}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T10:32:04.275294+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4977311", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "driver_error", "failureCode": "DRIVER_ERROR", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-21T18:39:29.750872+00:00", "modelId": "neuralmagic/Meta-Llama-3.1-8B-quantized.w8a8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 9084013360, "estimatedRequiredGiB": 10.162, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 9093204103, "modelscopeLicense": "llama3.1", "modelscopeParams": 8030261248, "modelscopeTags": ["license:llama3.1", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm", "custom_tag:quantized", "custom_tag:8-bit"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 9093204103}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T09:57:46.890719+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4976868", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDetectedFramework": "vllm", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "framework": "vllm", "lastSyncTime": "2026-09-20T10:06:20.452855+00:00", "modelId": "KoboldAI/fairseq-dense-13B-Janeway", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T09:55:21+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4079951", "taskType": "text-generation", "verifyResult": -1}
|
||||
@@ -254,6 +258,7 @@
|
||||
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["muse_glimmer"], "framework": "vllm", "lastSyncTime": "2026-09-21T02:15:17.867369+00:00", "modelId": "ddalcu/Muse-Glimmer-30B-MLX-Serve-4bit", "modelProfile": {"architectures": ["MuseGlimmerForConditionalGeneration"], "configFingerprint": "6db01b5037d5ed2db74875d2113f8582c97f34bd825637161e54174c9f2cabed", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 21407235494, "estimatedRequiredGiB": 23.956, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "muse_glimmer", "modelscopeFileSize": 21435726263, "modelscopeLicense": "apache-2.0", "modelscopeParams": 5792290816, "modelscopeTags": ["license:apache-2.0", "model_type:muse_glimmer", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:mlx-serve", "custom_tag:muse_glimmer"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 21435726263}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T04:02:52.182815+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4970446", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-21T08:44:09.587529+00:00", "modelId": "solidrust/Llama-3-8B-Instruct-v0.4-AWQ", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "6db01b5037d5ed2db74875d2113f8582c97f34bd825637161e54174c9f2cabed", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 5727938576, "estimatedRequiredGiB": 6.412, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 5737300809, "modelscopeLicense": "other", "modelscopeParams": 8030261248, "modelscopeTags": ["license:other", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:4-bit", "custom_tag:AWQ", "custom_tag:text-generation", "custom_tag:autotrain_compatible", "custom_tag:endpoints_compatible", "custom_tag:axolotl", "custom_tag:finetune", "custom_tag:facebook", "custom_tag:meta", "custom_tag:pytorch", "custom_tag:llama", "custom_tag:llama-3"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "awq", "repositoryOnDiskBytes": 5737300809}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T04:02:45.352471+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4970443", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-20T23:08:35.573468+00:00", "modelId": "solidrust/Llama-3-8B-Instruct-DPO-v0.1-AWQ", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "6db01b5037d5ed2db74875d2113f8582c97f34bd825637161e54174c9f2cabed", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 5727954960, "estimatedRequiredGiB": 6.412, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 5737317601, "modelscopeLicense": "other", "modelscopeParams": 8030269440, "modelscopeTags": ["license:other", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:4-bit", "custom_tag:AWQ", "custom_tag:text-generation", "custom_tag:autotrain_compatible", "custom_tag:endpoints_compatible", "custom_tag:axolotl", "custom_tag:finetune", "custom_tag:facebook", "custom_tag:meta", "custom_tag:pytorch", "custom_tag:llama", "custom_tag:llama-3"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "awq", "repositoryOnDiskBytes": 5737317601}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T04:02:45.350050+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4970442", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "driver_error", "failureCode": "DRIVER_ERROR", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-21T19:49:37.657097+00:00", "modelId": "neuralmagic/gemma-2-9b-it-quantized.w4a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7963207224, "estimatedRequiredGiB": 8.924, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 7985124714, "modelscopeLicense": "llama2", "modelscopeParams": 10159209984, "modelscopeTags": ["license:llama2", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 7985124714}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T04:02:45.348395+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4970436", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["zaya"], "framework": "vllm", "lastSyncTime": "2026-09-20T17:09:45.266103+00:00", "modelId": "JANGQ-AI/Osaurus-AppleScript-8B-JANG_6M", "modelProfile": {"architectures": ["ZayaForCausalLM"], "configFingerprint": "d341b8002cf39d3299fd429479e8c8e954583e49d4fec5a640ddc0f8f0cbd054", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7915599888, "estimatedRequiredGiB": 8.885, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "zaya", "modelscopeFileSize": 7950463197, "modelscopeLicense": "other", "modelscopeParams": 2274126328, "modelscopeTags": ["license:other", "model_type:zaya", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:applescript", "custom_tag:macos", "custom_tag:computer-use", "custom_tag:agent", "custom_tag:tool-calling", "custom_tag:function-calling", "custom_tag:mlx", "custom_tag:moe", "custom_tag:osaurus"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 7950463197}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T04:02:45.346171+00:00", "targetGpu": "Biren_166m", "taskId": "4970445", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-21T04:51:55.164906+00:00", "modelId": "OpenBMB/MiniCPM5-2B-GPTQ", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "6db01b5037d5ed2db74875d2113f8582c97f34bd825637161e54174c9f2cabed", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2099615880, "estimatedRequiredGiB": 2.358, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 2109841981, "modelscopeLicense": "apache-2.0", "modelscopeParams": 2516756480, "modelscopeTags": ["license:apache-2.0", "model_type:llama", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:minicpm", "custom_tag:minicpm5", "custom_tag:llama", "custom_tag:text-generation", "custom_tag:long-context", "custom_tag:tool-calling", "custom_tag:on-device", "custom_tag:edge-ai"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "awq", "repositoryOnDiskBytes": 2109841981}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T04:02:45.344449+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4970441", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-20T22:27:36.960842+00:00", "modelId": "solidrust/Llama-3-13B-Instruct-v0.1-AWQ", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "6db01b5037d5ed2db74875d2113f8582c97f34bd825637161e54174c9f2cabed", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 8447876488, "estimatedRequiredGiB": 9.452, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 8457284244, "modelscopeLicense": "other", "modelscopeParams": 13264949248, "modelscopeTags": ["license:other", "model_type:llama", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:4-bit", "custom_tag:AWQ", "custom_tag:text-generation", "custom_tag:autotrain_compatible", "custom_tag:endpoints_compatible", "custom_tag:mergekit", "custom_tag:merge", "custom_tag:facebook", "custom_tag:meta", "custom_tag:pytorch", "custom_tag:llama", "custom_tag:llama-3"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "awq", "repositoryOnDiskBytes": 8457284244}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T04:02:45.342540+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4970435", "taskType": "text-generation", "verifyResult": -1}
|
||||
@@ -273,6 +278,7 @@
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDetectedFramework": "vllm_fix_tokenizer", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-20T12:21:49.902097+00:00", "modelId": "XHToken/Spark-X2.5-4B", "modelProfile": {"architectures": ["Spark2_5ForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 8224192408, "estimatedRequiredGiB": 9.209, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "spark2_5", "modelscopeFileSize": 8239718268, "modelscopeLicense": "apache-2.0", "modelscopeParams": 4112079360, "modelscopeTags": ["license:apache-2.0", "model_type:spark2_5", "library:safetensors", "library:", "task:text-generation", "custom_tag:llm", "custom_tag:sparkx2_5"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 8239718268}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T03:58:35.457045+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4970357", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDetectedFramework": "vllm", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["gemma4"], "framework": "vllm_0_17_0_corex_4_4_0", "lastSyncTime": "2026-09-20T12:21:49.901805+00:00", "modelId": "mlx-community/gemma-4-26B-A4B-it-qat-OptiQ-4bit", "modelProfile": {"architectures": ["Gemma4ForConditionalGeneration"], "configFingerprint": "1cc16a7df03cf7b1485e2fda7e5c05e286a546d04e80bdd595c015004b3852a8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 21855018733, "estimatedRequiredGiB": 24.461, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma4", "modelscopeFileSize": 21887495988, "modelscopeLicense": "apache-2.0", "modelscopeParams": 6144703054, "modelscopeTags": ["license:apache-2.0", "model_type:gemma4", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:quantized", "custom_tag:mixed-precision", "custom_tag:4bit", "custom_tag:8bit", "custom_tag:optiq", "custom_tag:qat", "custom_tag:moe", "custom_tag:apple-silicon", "custom_tag:text-generation", "custom_tag:image-text-to-text", "custom_tag:gemma-4"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 21887495988}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T03:58:35.451794+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4970348", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "memory_capacity", "failureAction": "reject_if_estimated_load_exceeds_memory", "failureCategory": "memory_capacity", "failureClassificationReason": "structured_oom", "failureCode": "PREFLIGHT_OOM", "failureDetectedFramework": "vllm_fix_tokenizer", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureObservedGpuMemoryGiB": 32.0, "failureScope": "model_gpu", "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-20T12:21:49.901895+00:00", "modelId": "Youssofal/Qwen3.8-27B-MTPLX-Optimized-Speed", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 20683241105, "estimatedRequiredGiB": 23.138, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 20703487237, "modelscopeLicense": "apache-2.0", "modelscopeParams": 6086364400, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:apple-silicon", "custom_tag:macos", "custom_tag:speculative-decoding", "custom_tag:multi-token-prediction", "custom_tag:qwen", "custom_tag:qwen3.8", "custom_tag:mtp", "custom_tag:mtplx", "custom_tag:local-ai", "custom_tag:coding", "custom_tag:qwen3-8", "custom_tag:qwen-3.8", "custom_tag:local-llm", "custom_tag:llm", "custom_tag:m5", "custom_tag:m5-max", "custom_tag:m4", "custom_tag:m3", "custom_tag:macbook-pro", "custom_tag:mac-studio", "custom_tag:opencode", "custom_tag:claude-code", "custom_tag:27b", "custom_tag:qwen3.8-27b", "custom_tag:qwen3-8-27b"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 20703487237}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T03:58:35.449495+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4970352", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T19:49:37.657161+00:00", "modelId": "mlx-community/LFM2.5-1.2B-Instruct-4bit", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 658540250, "estimatedRequiredGiB": 0.741, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 663396475, "modelscopeLicense": "other", "modelscopeParams": 182975232, "modelscopeTags": ["license:other", "model_type:lfm2", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge", "custom_tag:mlx"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 663396475}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T03:58:35.441724+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4970354", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "driver_error", "failureCode": "DRIVER_ERROR", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-21T19:13:53.564431+00:00", "modelId": "RedHatAI/gemma-2-9b-it-quantized.w8a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 11998647728, "estimatedRequiredGiB": 13.434, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 12020563851, "modelscopeLicense": "gemma", "modelscopeParams": 10159209984, "modelscopeTags": ["license:gemma", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 12020563851}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T03:57:37.974956+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4970340", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "driver_error", "failureCode": "DRIVER_ERROR", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-21T03:25:53.878060+00:00", "modelId": "neuralmagic/gemma-2-2b-quantized.w8a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "a494b1667fab53609b94cc853db1a12c8d3b9fcf98795ecbab3dd5eb3a3c77f6", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 6746735136, "estimatedRequiredGiB": 7.565, "gpuMemoryEvidence": {"memoryGiB": 48.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 48.0, "maximumRepositorySizeGiB": 40.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 6768615496, "modelscopeLicense": "gemma", "modelscopeParams": 3204165888, "modelscopeTags": ["license:gemma", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 6768615496}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T03:57:37.934664+00:00", "targetGpu": "Mthreads_s4000", "taskId": "4970335", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "driver_error", "failureCode": "DRIVER_ERROR", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-21T10:57:20.777339+00:00", "modelId": "aisingapore/Gemma-SEA-LION-v4-27B-IT-NVFP4", "modelProfile": {"architectures": ["Gemma3ForConditionalGeneration"], "configFingerprint": "a494b1667fab53609b94cc853db1a12c8d3b9fcf98795ecbab3dd5eb3a3c77f6", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 20886763512, "estimatedRequiredGiB": 23.388, "gpuMemoryEvidence": {"memoryGiB": 48.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 48.0, "maximumRepositorySizeGiB": 40.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma3", "modelscopeFileSize": 20926892207, "modelscopeLicense": "gemma", "modelscopeParams": 28842037282, "modelscopeTags": ["license:gemma", "model_type:gemma3", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 20926892207}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T03:57:37.904873+00:00", "targetGpu": "Mthreads_s4000", "taskId": "4970336", "taskType": "text-generation", "verifyResult": -1}
|
||||
@@ -292,9 +298,3 @@
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-20T22:43:47.754882+00:00", "modelId": "IntervitensInc/kek_mk3", "modelProfile": {"architectures": ["StableLmForCausalLM"], "configFingerprint": "6db01b5037d5ed2db74875d2113f8582c97f34bd825637161e54174c9f2cabed", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 3289069288, "estimatedRequiredGiB": 3.683, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "stablelm", "modelscopeFileSize": 3295875324, "modelscopeLicense": null, "modelscopeParams": 1644515328, "modelscopeTags": ["model_type:stablelm", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 3295875324}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T03:46:35.813804+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4970120", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "missing_structured_error_code", "failureCode": null, "failureDetectedFramework": "transformers", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "transformers", "lastSyncTime": "2026-09-20T14:14:39.678925+00:00", "modelId": "aisingapore/SEA-LION-v1-7B", "modelProfile": {"architectures": ["MPTForCausalLM"], "configFingerprint": "0335c7e4668aaa963303ebe91f80d6eb1e7e8d0299011bd85f7e05a4df031dc7", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 15003361968, "estimatedRequiredGiB": 16.773, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "mpt", "modelscopeFileSize": 15008115847, "modelscopeLicense": "mit", "modelscopeParams": 7501651968, "modelscopeTags": ["license:mit", "model_type:mpt", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 15008115847}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T03:46:35.801267+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4970119", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-20T17:09:45.266093+00:00", "modelId": "aisingapore/Gemma-SEA-LION-v4-27B-IT-FP8-Dynamic", "modelProfile": {"architectures": ["Gemma3ForConditionalGeneration"], "configFingerprint": "d341b8002cf39d3299fd429479e8c8e954583e49d4fec5a640ddc0f8f0cbd054", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 29274355224, "estimatedRequiredGiB": 32.761, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma3", "modelscopeFileSize": 29314389092, "modelscopeLicense": "gemma", "modelscopeParams": 27432406640, "modelscopeTags": ["license:gemma", "model_type:gemma3", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 29314389092}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T03:46:35.789758+00:00", "targetGpu": "Biren_166m", "taskId": "4970118", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDetectedFramework": "vllm", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["zaya"], "framework": "vllm_0_17_0_corex_4_4_0", "lastSyncTime": "2026-09-20T12:01:00.671056+00:00", "modelId": "JANGQ-AI/Osaurus-AppleScript-8B-JANG_6M", "modelProfile": {"architectures": ["ZayaForCausalLM"], "configFingerprint": "1cc16a7df03cf7b1485e2fda7e5c05e286a546d04e80bdd595c015004b3852a8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7915599888, "estimatedRequiredGiB": 8.885, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "zaya", "modelscopeFileSize": 7950463197, "modelscopeLicense": "other", "modelscopeParams": 2274126328, "modelscopeTags": ["license:other", "model_type:zaya", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:applescript", "custom_tag:macos", "custom_tag:computer-use", "custom_tag:agent", "custom_tag:tool-calling", "custom_tag:function-calling", "custom_tag:mlx", "custom_tag:moe", "custom_tag:osaurus"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 7950463197}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T03:46:35.783924+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4970117", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDetectedFramework": "vllm", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["muse_glimmer"], "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-21T13:58:33.470413+00:00", "modelId": "ddalcu/Muse-Glimmer-30B-MLX-Serve-4bit", "modelProfile": {"architectures": ["MuseGlimmerForConditionalGeneration"], "configFingerprint": "1c0a4f69628b2a0ad37a6eee139a4c98afed7d8c93529a9ef42a90756e4e1940", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 21407235494, "estimatedRequiredGiB": 23.956, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "muse_glimmer", "modelscopeFileSize": 21435726263, "modelscopeLicense": "apache-2.0", "modelscopeParams": 5792290816, "modelscopeTags": ["license:apache-2.0", "model_type:muse_glimmer", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:mlx-serve", "custom_tag:muse_glimmer"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 21435726263}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T03:46:29.643802+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4970111", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_0_17_0_corex_4_4_0", "lastSyncTime": "2026-09-20T12:01:00.670830+00:00", "modelId": "mlx-community/KAT-Coder-V2.5-Dev-OptiQ-4bit", "modelProfile": {"architectures": ["Qwen3_5MoeForConditionalGeneration"], "configFingerprint": "1cc16a7df03cf7b1485e2fda7e5c05e286a546d04e80bdd595c015004b3852a8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 21930054875, "estimatedRequiredGiB": 24.532, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_moe", "modelscopeFileSize": 21950415614, "modelscopeLicense": "apache-2.0", "modelscopeParams": 6024588416, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5_moe", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:optiq", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 21950415614}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T03:46:29.635942+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4970115", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["qwen3_5"], "framework": "vllm-customized", "lastSyncTime": "2026-09-20T22:27:36.960800+00:00", "modelId": "Mia-AiLab/Qwen3.8-27B-EXL3-3.5bpw", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 15338408461, "estimatedRequiredGiB": 17.168, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 15362064483, "modelscopeLicense": "apache-2.0", "modelscopeParams": 7669052656, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:exl3", "custom_tag:exllamav3", "custom_tag:quantization", "custom_tag:long-context", "custom_tag:speculative-decoding"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "exl3", "repositoryOnDiskBytes": 15362064483}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T03:41:49.651764+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4970045", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["qwen3_5"], "framework": "vllm-customized", "lastSyncTime": "2026-09-21T04:51:55.164962+00:00", "modelId": "ornith-ai/Ornith-1.5-9B-NVFP4", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 8818103892, "estimatedRequiredGiB": 9.883, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 8842796317, "modelscopeLicense": "mit", "modelscopeParams": 6728625904, "modelscopeTags": ["license:mit", "model_type:qwen3_5", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "modelopt", "repositoryOnDiskBytes": 8842796317}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T03:41:49.636924+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4970040", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDetectedFramework": "vllm", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["gemma4"], "framework": "vllm-patch-tokenizer", "lastSyncTime": "2026-09-21T14:32:55.759813+00:00", "modelId": "mlx-community/gemma-4-e2b-it-OptiQ-4bit", "modelProfile": {"architectures": ["Gemma4ForConditionalGeneration"], "configFingerprint": "0d5bc82787a89d4313672808d1a8629a1fc675c91aa6a08865f770aa24447fbc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 5226595850, "estimatedRequiredGiB": 5.878, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma4", "modelscopeFileSize": 5259113427, "modelscopeLicense": "gemma", "modelscopeParams": 1141165347, "modelscopeTags": ["license:gemma", "model_type:gemma4", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:quantized", "custom_tag:mixed-precision", "custom_tag:4bit", "custom_tag:8bit", "custom_tag:optiq", "custom_tag:apple-silicon", "custom_tag:text-generation", "custom_tag:gemma-4"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 5259113427}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T03:41:49.472121+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4970033", "taskType": "text-generation", "verifyResult": -1}
|
||||
|
||||
Reference in New Issue
Block a user