state: generation 11344 (intent)

This commit is contained in:
2026-09-21 08:46:03 +00:00
parent 1068341339
commit a9635fd320
8 changed files with 2423 additions and 2305 deletions

View File

@@ -1784,7 +1784,7 @@
"kunlunxin_p-800|vllm_tokenizer_patch|text-generation|model_type:qwen3_5_moe": {
"architectureSignature": "model_type:qwen3_5_moe",
"architectures": [],
"evidenceCount": 2,
"evidenceCount": 1,
"expiresAt": "2026-10-20T03:09:17.939093+00:00",
"framework": "vllm_tokenizer_patch",
"latestFailureAt": "2026-09-20T03:09:17.939093+00:00",
@@ -1792,12 +1792,10 @@
"matchType": "model_type",
"modelType": "qwen3_5_moe",
"sourceModelIds": [
"mlx-community/XYZ-Aquila-mini-OptiQ-4bit",
"apodex/Apodex-1.1-mini-GPTQ-Int4"
"mlx-community/XYZ-Aquila-mini-OptiQ-4bit"
],
"sourceTaskIds": [
"4969329",
"4969320"
"4969329"
],
"targetGpu": "Kunlunxin_p-800",
"taskType": "text-generation"
@@ -2656,7 +2654,7 @@
"taskType": "text-generation"
}
},
"generatedAt": "2026-09-21T08:25:33.731333+00:00",
"generatedAt": "2026-09-21T08:44:12.237609+00:00",
"summary": {
"activeBlockCount": 133,
"byGpuFramework": {

View File

@@ -425,7 +425,7 @@
}
},
"frameworkUpdatedAt": null,
"generatedAt": "2026-09-21T08:43:07.573772+00:00",
"generatedAt": "2026-09-21T08:44:20.369063+00:00",
"gpuStats": {
"Ascend_910-b3": {
"available": true,

File diff suppressed because it is too large Load Diff

File diff suppressed because it is too large Load Diff

View File

@@ -159,8 +159,10 @@
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_0_17_0_corex_4_4_0", "lastSyncTime": "2026-09-20T12:46:14.462946+00:00", "modelId": "mlx-community/Devstral-Small-2-24B-Instruct-2512-OptiQ-4bit", "modelProfile": {"architectures": ["Mistral3ForConditionalGeneration"], "configFingerprint": "1cc16a7df03cf7b1485e2fda7e5c05e286a546d04e80bdd595c015004b3852a8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 17367755740, "estimatedRequiredGiB": 19.429, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "mistral3", "modelscopeFileSize": 17385072379, "modelscopeLicense": "apache-2.0", "modelscopeParams": 4929899520, "modelscopeTags": ["license:apache-2.0", "model_type:mistral3", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:quantized", "custom_tag:mixed-precision", "custom_tag:4bit", "custom_tag:8bit", "custom_tag:optiq", "custom_tag:apple-silicon", "custom_tag:text-generation", "custom_tag:mistral", "custom_tag:mistral3", "custom_tag:devstral"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 17385072379}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T04:26:45.153348+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4970732", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDetectedFramework": "vllm", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["gemma4_unified"], "framework": "vllm_0_17_0_corex_4_4_0", "lastSyncTime": "2026-09-20T12:46:14.462967+00:00", "modelId": "mlx-community/gemma-4-12B-it-qat-OptiQ-4bit", "modelProfile": {"architectures": ["Gemma4UnifiedForConditionalGeneration"], "configFingerprint": "1cc16a7df03cf7b1485e2fda7e5c05e286a546d04e80bdd595c015004b3852a8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 9003966484, "estimatedRequiredGiB": 10.099, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma4_unified", "modelscopeFileSize": 9036402901, "modelscopeLicense": "apache-2.0", "modelscopeParams": 2463563824, "modelscopeTags": ["license:apache-2.0", "model_type:gemma4_unified", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:quantized", "custom_tag:mixed-precision", "custom_tag:4bit", "custom_tag:8bit", "custom_tag:optiq", "custom_tag:qat", "custom_tag:apple-silicon", "custom_tag:text-generation", "custom_tag:image-text-to-text", "custom_tag:gemma-4"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 9036402901}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T04:26:45.138158+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4970728", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "参数/模板问题", "framework": "vllm-customized", "lastSyncTime": "2026-09-21T06:41:38.663980+00:00", "modelId": "Youssofal/Qwen3.8-27B-MTPLX-Optimized-Speed", "modelProfile": {}, "outcome": "failed", "status": "failed", "submitTime": "2026-09-20T04:15:36.979619+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4970617", "taskType": "text-generation", "verifyResult": null}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm-mlu", "lastSyncTime": "2026-09-21T08:44:09.587559+00:00", "modelId": "mlx-community/LFM2.5-1.2B-Instruct-4bit", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 658540250, "estimatedRequiredGiB": 0.741, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 663396475, "modelscopeLicense": "other", "modelscopeParams": 182975232, "modelscopeTags": ["license:other", "model_type:lfm2", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge", "custom_tag:mlx"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 663396475}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T04:15:36.970472+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4970618", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_0_17_0_corex_4_4_0", "lastSyncTime": "2026-09-20T12:34:29.856403+00:00", "modelId": "Youssofal/Qwen3.5-9B-MTPLX-Optimized-Speed", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "1cc16a7df03cf7b1485e2fda7e5c05e286a546d04e80bdd595c015004b3852a8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 8674988799, "estimatedRequiredGiB": 9.718, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 8695119291, "modelscopeLicense": "apache-2.0", "modelscopeParams": 2415484144, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:apple-silicon", "custom_tag:speculative-decoding", "custom_tag:qwen", "custom_tag:qwen3", "custom_tag:mtp", "custom_tag:mtplx", "custom_tag:local-ai", "custom_tag:qwen3-5", "custom_tag:9b", "custom_tag:6-bit"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 8695119291}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T04:15:36.958375+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4970614", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDetectedFramework": "transformers", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "framework": "transformers", "lastSyncTime": "2026-09-20T12:34:29.856377+00:00", "modelId": "BAAI/RoboBrain2.5-8B-MT", "modelProfile": {"architectures": ["Qwen3VLForConditionalGeneration"], "configFingerprint": "3086a71adcdca90005b4967eb30749c44cd82389aff8d599758e4b91675d40dc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 17534339552, "estimatedRequiredGiB": 19.609, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_vl", "modelscopeFileSize": 17545918174, "modelscopeLicense": "apache-2.0", "modelscopeParams": 8767123696, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_vl", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 17545918174}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T04:15:36.954828+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4970615", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm-customized", "lastSyncTime": "2026-09-21T08:44:09.587595+00:00", "modelId": "LiquidAI/LFM2.5-1.2B-Thinking-MLX-8bit", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 1243645809, "estimatedRequiredGiB": 1.395, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 1248516007, "modelscopeLicense": "other", "modelscopeParams": 329251584, "modelscopeTags": ["license:other", "model_type:lfm2", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge", "custom_tag:mlx", "custom_tag:reasoning"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 1248516007}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T04:14:42.077355+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4970592", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm-customized", "lastSyncTime": "2026-09-21T02:15:17.867294+00:00", "modelId": "LiquidAI/LFM2.5-1.2B-Instruct-MLX-bf16", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2340697867, "estimatedRequiredGiB": 2.622, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 2345693224, "modelscopeLicense": "other", "modelscopeParams": 1170340608, "modelscopeTags": ["license:other", "model_type:lfm2", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge", "custom_tag:mlx"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 2345693224}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T04:14:42.074603+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4970596", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "model_load", "failureAction": "inspect_root_exception_then_llm_if_unknown", "failureCategory": "model_load", "failureClassificationReason": "broad_load_error", "failureCode": "MODEL_LOAD_FAILED", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "model_gpu_framework", "framework": "vllm", "lastSyncTime": "2026-09-21T04:51:55.164941+00:00", "modelId": "aisingapore/Gemma-SEA-LION-v4-27B-IT-NVFP4", "modelProfile": {"architectures": ["Gemma3ForConditionalGeneration"], "configFingerprint": "533968aac3b9c90272dfbe2866791f782e077f0dceee5b6ffffce48ba6674a10", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 20886763512, "estimatedRequiredGiB": 23.388, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma3", "modelscopeFileSize": 20926892207, "modelscopeLicense": "gemma", "modelscopeParams": 28842037282, "modelscopeTags": ["license:gemma", "model_type:gemma3", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 20926892207}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T04:14:42.066492+00:00", "targetGpu": "MetaX_c-500", "taskId": "4970593", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDetectedFramework": "vllm_fix_tokenizer", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-20T12:34:29.856389+00:00", "modelId": "BAAI/AquilaMed-RL", "modelProfile": {"architectures": ["AquilaForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 6968, "estimatedRequiredGiB": 18.386, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "aquila3", "modelscopeFileSize": 16451477782, "modelscopeLicense": "other", "modelscopeParams": 8223748096, "modelscopeTags": ["license:other", "model_type:aquila3", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:arxiv:2406.12182"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 16451477782}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T04:14:41.855366+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4970587", "taskType": "text-generation", "verifyResult": -1}
@@ -176,6 +178,7 @@
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_0_17_0_corex_4_4_0", "lastSyncTime": "2026-09-20T12:34:29.856340+00:00", "modelId": "Arain119/Sophia", "modelProfile": {"architectures": ["SophiaForCausalLM"], "configFingerprint": "1cc16a7df03cf7b1485e2fda7e5c05e286a546d04e80bdd595c015004b3852a8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2226652224, "estimatedRequiredGiB": 7.28, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "sophia_hybrid", "modelscopeFileSize": 6513869659, "modelscopeLicense": "Apache License 2.0", "modelscopeParams": 1113293776, "modelscopeTags": ["license:Apache License 2.0", "model_type:sophia_hybrid", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 6513869659}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T04:09:56.493446+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4970510", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["qwen3"], "framework": "vllm", "lastSyncTime": "2026-09-20T22:24:01.251292+00:00", "modelId": "whcl412/mlx-LycheeAI-coder-1.7b", "modelProfile": {"architectures": ["Qwen3ForCausalLM"], "configFingerprint": "6db01b5037d5ed2db74875d2113f8582c97f34bd825637161e54174c9f2cabed", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 968080210, "estimatedRequiredGiB": 1.095, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3", "modelscopeFileSize": 979581804, "modelscopeLicense": "apache-2.0", "modelscopeParams": 268944384, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3", "library:lora", "library:safetensors", "task:text-generation", "custom_tag:qwen3", "custom_tag:coder", "custom_tag:lora", "custom_tag:code", "custom_tag:lychee", "custom_tag:finetune", "custom_tag:multilingual"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 979581804}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T04:02:52.200512+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4970447", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["muse_glimmer"], "framework": "vllm", "lastSyncTime": "2026-09-21T02:15:17.867369+00:00", "modelId": "ddalcu/Muse-Glimmer-30B-MLX-Serve-4bit", "modelProfile": {"architectures": ["MuseGlimmerForConditionalGeneration"], "configFingerprint": "6db01b5037d5ed2db74875d2113f8582c97f34bd825637161e54174c9f2cabed", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 21407235494, "estimatedRequiredGiB": 23.956, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "muse_glimmer", "modelscopeFileSize": 21435726263, "modelscopeLicense": "apache-2.0", "modelscopeParams": 5792290816, "modelscopeTags": ["license:apache-2.0", "model_type:muse_glimmer", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:mlx-serve", "custom_tag:muse_glimmer"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 21435726263}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T04:02:52.182815+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4970446", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-21T08:44:09.587529+00:00", "modelId": "solidrust/Llama-3-8B-Instruct-v0.4-AWQ", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "6db01b5037d5ed2db74875d2113f8582c97f34bd825637161e54174c9f2cabed", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 5727938576, "estimatedRequiredGiB": 6.412, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 5737300809, "modelscopeLicense": "other", "modelscopeParams": 8030261248, "modelscopeTags": ["license:other", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:4-bit", "custom_tag:AWQ", "custom_tag:text-generation", "custom_tag:autotrain_compatible", "custom_tag:endpoints_compatible", "custom_tag:axolotl", "custom_tag:finetune", "custom_tag:facebook", "custom_tag:meta", "custom_tag:pytorch", "custom_tag:llama", "custom_tag:llama-3"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "awq", "repositoryOnDiskBytes": 5737300809}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T04:02:45.352471+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4970443", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-20T23:08:35.573468+00:00", "modelId": "solidrust/Llama-3-8B-Instruct-DPO-v0.1-AWQ", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "6db01b5037d5ed2db74875d2113f8582c97f34bd825637161e54174c9f2cabed", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 5727954960, "estimatedRequiredGiB": 6.412, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 5737317601, "modelscopeLicense": "other", "modelscopeParams": 8030269440, "modelscopeTags": ["license:other", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:4-bit", "custom_tag:AWQ", "custom_tag:text-generation", "custom_tag:autotrain_compatible", "custom_tag:endpoints_compatible", "custom_tag:axolotl", "custom_tag:finetune", "custom_tag:facebook", "custom_tag:meta", "custom_tag:pytorch", "custom_tag:llama", "custom_tag:llama-3"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "awq", "repositoryOnDiskBytes": 5737317601}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T04:02:45.350050+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4970442", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["zaya"], "framework": "vllm", "lastSyncTime": "2026-09-20T17:09:45.266103+00:00", "modelId": "JANGQ-AI/Osaurus-AppleScript-8B-JANG_6M", "modelProfile": {"architectures": ["ZayaForCausalLM"], "configFingerprint": "d341b8002cf39d3299fd429479e8c8e954583e49d4fec5a640ddc0f8f0cbd054", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7915599888, "estimatedRequiredGiB": 8.885, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "zaya", "modelscopeFileSize": 7950463197, "modelscopeLicense": "other", "modelscopeParams": 2274126328, "modelscopeTags": ["license:other", "model_type:zaya", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:applescript", "custom_tag:macos", "custom_tag:computer-use", "custom_tag:agent", "custom_tag:tool-calling", "custom_tag:function-calling", "custom_tag:mlx", "custom_tag:moe", "custom_tag:osaurus"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 7950463197}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T04:02:45.346171+00:00", "targetGpu": "Biren_166m", "taskId": "4970445", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-21T04:51:55.164906+00:00", "modelId": "OpenBMB/MiniCPM5-2B-GPTQ", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "6db01b5037d5ed2db74875d2113f8582c97f34bd825637161e54174c9f2cabed", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2099615880, "estimatedRequiredGiB": 2.358, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 2109841981, "modelscopeLicense": "apache-2.0", "modelscopeParams": 2516756480, "modelscopeTags": ["license:apache-2.0", "model_type:llama", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:minicpm", "custom_tag:minicpm5", "custom_tag:llama", "custom_tag:text-generation", "custom_tag:long-context", "custom_tag:tool-calling", "custom_tag:on-device", "custom_tag:edge-ai"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "awq", "repositoryOnDiskBytes": 2109841981}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T04:02:45.344449+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4970441", "taskType": "text-generation", "verifyResult": -1}
@@ -231,6 +234,7 @@
{"failReason": "参数/模板问题", "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-21T02:15:17.867223+00:00", "modelId": "Youssofal/Qwen3.8-27B-MTPLX-Bare-Speed-FP16", "modelProfile": {}, "outcome": "failed", "status": "failed", "submitTime": "2026-09-20T03:41:48.960695+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4970000", "taskType": "text-generation", "verifyResult": null}
{"failReason": "platform_infrastructure", "failureAction": "open_short_gpu_framework_circuit_and_retry_other_models", "failureCategory": "platform_infrastructure", "failureClassificationReason": "platform_io_transient", "failureCode": "STORAGE_ERROR", "failureDetectedFramework": "vllm", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "gpu_framework", "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-21T08:17:25.672552+00:00", "modelId": "RedHatAI/Qwen2-1.5B-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["Qwen2ForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2245331488, "estimatedRequiredGiB": 2.522, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen2", "modelscopeFileSize": 2256941144, "modelscopeLicense": "apache-2.0", "modelscopeParams": 1777088000, "modelscopeTags": ["license:apache-2.0", "model_type:qwen2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 2256941144}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T03:41:42.246689+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4969991", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDetectedFramework": "vllm", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["qwen3_5"], "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T03:29:39.969809+00:00", "modelId": "Youssofal/Qwen3.5-9B-MTPLX-Optimized-Speed", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "cbe5569cdb6c42553436b837dd5b8f0cfa03334642a56f4ff0137f952f8eb84b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 8674988799, "estimatedRequiredGiB": 9.718, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 8695119291, "modelscopeLicense": "apache-2.0", "modelscopeParams": 2415484144, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:apple-silicon", "custom_tag:speculative-decoding", "custom_tag:qwen", "custom_tag:qwen3", "custom_tag:mtp", "custom_tag:mtplx", "custom_tag:local-ai", "custom_tag:qwen3-5", "custom_tag:9b", "custom_tag:6-bit"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 8695119291}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T03:41:42.245581+00:00", "targetGpu": "Vastai_va16", "taskId": "4969995", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T08:44:09.587569+00:00", "modelId": "RedHatAI/Meta-Llama-3.1-8B-Instruct-FP8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 9081287016, "estimatedRequiredGiB": 10.159, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 9090504850, "modelscopeLicense": "llama3.1", "modelscopeParams": 8030261248, "modelscopeTags": ["license:llama3.1", "model_type:llama", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:fp8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 9090504850}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T03:40:37.335603+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4969955", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "tokenizer_compatibility", "failureAction": "check_tokenizer_files_then_llm", "failureCategory": "tokenizer_compatibility", "failureClassificationReason": "broad_tokenizer_error", "failureCode": "TOKENIZER_FAILED", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "model_framework", "framework": "vllm", "lastSyncTime": "2026-09-21T08:17:25.672573+00:00", "modelId": "OpenBMB/MiniCPM5-2B-MLX", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 1416035216, "estimatedRequiredGiB": 1.594, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 1426233716, "modelscopeLicense": "apache-2.0", "modelscopeParams": 393390080, "modelscopeTags": ["license:apache-2.0", "model_type:llama", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:minicpm", "custom_tag:minicpm5", "custom_tag:llama", "custom_tag:text-generation", "custom_tag:long-context", "custom_tag:tool-calling", "custom_tag:on-device", "custom_tag:edge-ai"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 1426233716}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T03:30:31.102178+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4969810", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "driver_error", "failureCode": "DRIVER_ERROR", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-21T00:42:10.765065+00:00", "modelId": "OpenBMB/MiniCPM5-2B-GPTQ", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2099615880, "estimatedRequiredGiB": 2.358, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 2109841981, "modelscopeLicense": "apache-2.0", "modelscopeParams": 2516756480, "modelscopeTags": ["license:apache-2.0", "model_type:llama", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:minicpm", "custom_tag:minicpm5", "custom_tag:llama", "custom_tag:text-generation", "custom_tag:long-context", "custom_tag:tool-calling", "custom_tag:on-device", "custom_tag:edge-ai"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "awq", "repositoryOnDiskBytes": 2109841981}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T03:30:31.100436+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4969813", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "driver_error", "failureCode": "DRIVER_ERROR", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-21T08:17:25.672451+00:00", "modelId": "aisingapore/SEA-LION-v1-7B", "modelProfile": {"architectures": ["MPTForCausalLM"], "configFingerprint": "58bd440e460cf3e4d9271b3c32c451e80254e59faf47ce55dcda11588cf4117e", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 15003361968, "estimatedRequiredGiB": 16.773, "gpuMemoryEvidence": {"memoryGiB": 48.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 48.0, "maximumRepositorySizeGiB": 40.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "mpt", "modelscopeFileSize": 15008115847, "modelscopeLicense": "mit", "modelscopeParams": 7501651968, "modelscopeTags": ["license:mit", "model_type:mpt", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 15008115847}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T03:30:23.765553+00:00", "targetGpu": "Mthreads_s4000", "taskId": "4969800", "taskType": "text-generation", "verifyResult": -1}
@@ -294,7 +298,3 @@
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDetectedFramework": "vllm", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["nemotron_h"], "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-21T03:52:45.758119+00:00", "modelId": "primitive-ai/Nemotron-3.5-Lightning-30B-A3B-mixed-INT4-INT8", "modelProfile": {"architectures": ["NemotronHForCausalLM"], "configFingerprint": "1c0a4f69628b2a0ad37a6eee139a4c98afed7d8c93529a9ef42a90756e4e1940", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 20628596944, "estimatedRequiredGiB": 23.076, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "nemotron_h", "modelscopeFileSize": 20647674585, "modelscopeLicense": "other", "modelscopeParams": 33943909952, "modelscopeTags": ["license:other", "model_type:nemotron_h", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:vllm", "custom_tag:compressed-tensors", "custom_tag:int4", "custom_tag:int8", "custom_tag:mixed-precision", "custom_tag:quantized", "custom_tag:moe", "custom_tag:mamba", "custom_tag:blackwell"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 20647674585}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T03:09:17.944560+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4969327", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "参数/模板问题", "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-21T02:15:17.867234+00:00", "modelId": "ewinregirgojr/Qwen3.8-14B-Instruct-Turbo", "modelProfile": {}, "outcome": "failed", "status": "failed", "submitTime": "2026-09-20T03:09:17.940876+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4969328", "taskType": "text-generation", "verifyResult": null}
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDetectedFramework": "vllm", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["qwen3_5_moe"], "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-21T03:45:50.542852+00:00", "modelId": "mlx-community/XYZ-Aquila-mini-OptiQ-4bit", "modelProfile": {"architectures": ["Qwen3_5MoeForConditionalGeneration"], "configFingerprint": "1c0a4f69628b2a0ad37a6eee139a4c98afed7d8c93529a9ef42a90756e4e1940", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 37184031394, "estimatedRequiredGiB": 41.579, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_moe", "modelscopeFileSize": 37204398600, "modelscopeLicense": "apache-2.0", "modelscopeParams": 10061358960, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5_moe", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:optiq", "custom_tag:mixed-precision", "custom_tag:qwen3.6", "custom_tag:agentic-search"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 37204398600}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T03:09:17.939093+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4969329", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "参数/模板问题", "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-21T02:15:17.867207+00:00", "modelId": "ewinregirgojr/Qwen3.8-9B-Instruct-Turbo", "modelProfile": {}, "outcome": "failed", "status": "failed", "submitTime": "2026-09-20T03:09:17.880383+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4969326", "taskType": "text-generation", "verifyResult": null}
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDetectedFramework": "vllm", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["qwen3_5_text"], "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-21T03:45:50.542829+00:00", "modelId": "TokenRhythm/NeoHorse-1-4B", "modelProfile": {"architectures": ["Qwen3_5ForCausalLM"], "configFingerprint": "1c0a4f69628b2a0ad37a6eee139a4c98afed7d8c93529a9ef42a90756e4e1940", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 8411552560, "estimatedRequiredGiB": 9.428, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_text", "modelscopeFileSize": 8435794426, "modelscopeLicense": "Apache License 2.0", "modelscopeParams": 4205751296, "modelscopeTags": ["license:Apache License 2.0", "model_type:qwen3_5_text", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:agentic", "custom_tag:tool-use", "custom_tag:coding", "custom_tag:reasoning", "custom_tag:instruction-following"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 8435794426}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T03:09:17.849368+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4969324", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDetectedFramework": "vllm", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["qwen3_5_moe"], "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T20:49:19.683330+00:00", "modelId": "apodex/Apodex-1.1-mini-GPTQ-Int4", "modelProfile": {"architectures": ["Qwen3_5MoeForConditionalGeneration"], "configFingerprint": "1c0a4f69628b2a0ad37a6eee139a4c98afed7d8c93529a9ef42a90756e4e1940", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 24651300904, "estimatedRequiredGiB": 27.587, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_moe", "modelscopeFileSize": 24684544516, "modelscopeLicense": "apache-2.0", "modelscopeParams": 35951822704, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5_moe", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:agent", "custom_tag:apodex", "custom_tag:arxiv:2608.23283"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "gptq", "repositoryOnDiskBytes": 24684544516}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T03:09:17.835766+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4969320", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "参数/模板问题", "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-21T02:15:17.867242+00:00", "modelId": "ornith-ai/Ornith-1.5-9B-NVFP4", "modelProfile": {}, "outcome": "failed", "status": "failed", "submitTime": "2026-09-20T03:09:17.795054+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4969322", "taskType": "text-generation", "verifyResult": null}

View File

@@ -2018,6 +2018,62 @@
{"batchId": "f1d22b7d76084321a5e6a45d08bf0104", "completedAt": "2026-09-21T08:42:21.948160+00:00", "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:36:59.735422+00:00", "framework": "vllm", "intentId": "35a01ff1dc3a4205b731387db2515c0a", "lastModified": "2026-08-24T20:04:01+00:00", "modelAddress": "https://modelscope.cn/models/neuralmagic/SmolLM-1.7B-Instruct-quantized.w8a16", "reason": null, "reconciledAt": "2026-09-21T08:43:04.362550+00:00", "repoId": "neuralmagic/SmolLM-1.7B-Instruct-quantized.w8a16", "safeConfigVector": {"dtype": "-", "gpuNum": 1, "maxModelLen": 2048}, "status": "recovered_active", "targetGpu": "Ascend_910-b4", "taskId": "4997284", "taskType": "text-generation"}
{"batchId": "5464b6ee30614a02be7c20a3cd2e0073", "completedAt": "2026-09-21T08:43:02.235660+00:00", "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:42:52.873803+00:00", "framework": "vllm_fix_tokenizer", "intentId": "36d9583802194b358a522a2738b0698b", "lastModified": "2026-09-21T01:29:45+00:00", "modelAddress": "https://modelscope.cn/models/blue2star/Qwen-Image-2.1-PE-I2I-ComfyUI", "reason": null, "reconciledAt": "2026-09-21T08:43:04.362245+00:00", "repoId": "blue2star/Qwen-Image-2.1-PE-I2I-ComfyUI", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "recovered_active", "targetGpu": "Kunlunxin_p-800", "taskId": "4997305", "taskType": "text-generation"}
{"batchId": "5464b6ee30614a02be7c20a3cd2e0073", "completedAt": "2026-09-21T08:43:02.235686+00:00", "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:42:52.873924+00:00", "framework": "vllm_fix_tokenizer", "intentId": "1e28fad645484a6c88a3caad773fd67e", "lastModified": "2026-09-17T13:06:10+00:00", "modelAddress": "https://modelscope.cn/models/Youssofal/Qwen3.8-27B-MTPLX-Optimized-Quality", "reason": null, "reconciledAt": "2026-09-21T08:43:04.363264+00:00", "repoId": "Youssofal/Qwen3.8-27B-MTPLX-Optimized-Quality", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "recovered_active", "targetGpu": "Biren_166m", "taskId": "4997304", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.901388+00:00", "framework": "vllm_tokenizer_patch", "intentId": "e25e8f70e9444a82a5030a7a0cf3c216", "lastModified": "2026-08-24T18:24:11+00:00", "modelAddress": "https://modelscope.cn/models/nm-testing/tinyllama-oneshot-w8a8-dynamic-token-v2", "repoId": "nm-testing/tinyllama-oneshot-w8a8-dynamic-token-v2", "safeConfigVector": {"gpuNum": 1}, "status": "pending", "targetGpu": "Ascend_910-b4", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.901513+00:00", "framework": "vllm_tokenizer_patch", "intentId": "c47ea83734b5464ebb691c8af83e5ce7", "lastModified": "2026-08-24T19:10:38+00:00", "modelAddress": "https://modelscope.cn/models/RedHatAI/SmolLM-135M-Instruct-quantized.w8a16", "repoId": "RedHatAI/SmolLM-135M-Instruct-quantized.w8a16", "safeConfigVector": {"gpuNum": 1}, "status": "pending", "targetGpu": "Ascend_910-b4", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "36ec94e289a8da5fd70b98a2f401fc5159740daf9e2a076da4fd490f34bcbc0b", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.901571+00:00", "framework": "vllm-mlu", "intentId": "01a049abe8b241b091eab320182370c6", "lastModified": "2026-08-24T19:21:33+00:00", "modelAddress": "https://modelscope.cn/models/neuralmagic/starcoder2-3b-quantized.w8a8", "repoId": "neuralmagic/starcoder2-3b-quantized.w8a8", "safeConfigVector": {"dtype": "-", "gpuNum": 1}, "status": "pending", "targetGpu": "Cambricon_mlu-370-x4", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "b3ef6bcf57e6bb4fc9ede989e67a765b0e40b19fbd407f1f8452833c12bd5c38", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.901629+00:00", "framework": "vllm-mlu", "intentId": "44ba954dd43e47dc88a129b8397413ed", "lastModified": "2026-08-24T18:26:53+00:00", "modelAddress": "https://modelscope.cn/models/nm-testing/tinyllama-oneshot-w8a16-per-channel", "repoId": "nm-testing/tinyllama-oneshot-w8a16-per-channel", "safeConfigVector": {"dtype": "-", "gpuNum": 1}, "status": "pending", "targetGpu": "Cambricon_mlu-370-x4", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "36ec94e289a8da5fd70b98a2f401fc5159740daf9e2a076da4fd490f34bcbc0b", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.901693+00:00", "framework": "vllm-mlu", "intentId": "320214167d5e48f2ba79162fa9996915", "lastModified": "2026-08-24T19:34:42+00:00", "modelAddress": "https://modelscope.cn/models/RedHatAI/Phi-3-mini-128k-instruct-FP8", "repoId": "RedHatAI/Phi-3-mini-128k-instruct-FP8", "safeConfigVector": {"dtype": "-", "gpuNum": 1}, "status": "pending", "targetGpu": "Cambricon_mlu-370-x4", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "36ec94e289a8da5fd70b98a2f401fc5159740daf9e2a076da4fd490f34bcbc0b", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.901746+00:00", "framework": "vllm-mlu", "intentId": "70f8a6348edd410ca8491ebb341261b6", "lastModified": "2026-08-24T19:08:49+00:00", "modelAddress": "https://modelscope.cn/models/RedHatAI/Mistral-Nemo-Instruct-2407-quantized.w8a16", "repoId": "RedHatAI/Mistral-Nemo-Instruct-2407-quantized.w8a16", "safeConfigVector": {"dtype": "-", "gpuNum": 1}, "status": "pending", "targetGpu": "Cambricon_mlu-370-x4", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "b3ef6bcf57e6bb4fc9ede989e67a765b0e40b19fbd407f1f8452833c12bd5c38", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.901799+00:00", "framework": "vllm-mlu", "intentId": "e353966edf5e439c9a17e9e79eb68b96", "lastModified": "2026-08-24T19:35:49+00:00", "modelAddress": "https://modelscope.cn/models/RedHatAI/SmolLM-1.7B-Instruct-quantized.w8a16", "repoId": "RedHatAI/SmolLM-1.7B-Instruct-quantized.w8a16", "safeConfigVector": {"dtype": "-", "gpuNum": 1}, "status": "pending", "targetGpu": "Cambricon_mlu-370-x4", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "c25be7d7c1bf173abe564916f8a370d9b648beb3aea2a79522523aaa6b69ee29", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.901851+00:00", "framework": "vllm", "intentId": "d7da673ed48442919b786bedc9ad1afc", "lastModified": "2026-08-24T18:26:39+00:00", "modelAddress": "https://modelscope.cn/models/nm-testing/tinyllama-oneshot-w8a8-static-v2", "repoId": "nm-testing/tinyllama-oneshot-w8a8-static-v2", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 2048}, "status": "pending", "targetGpu": "Ascend_910-b3", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "c25be7d7c1bf173abe564916f8a370d9b648beb3aea2a79522523aaa6b69ee29", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.901903+00:00", "framework": "vllm", "intentId": "cab26704dbaf4f13bb40dfb19e660cec", "lastModified": "2026-08-24T18:28:40+00:00", "modelAddress": "https://modelscope.cn/models/nm-testing/tinyllama-marlin24-w4a16-group128", "repoId": "nm-testing/tinyllama-marlin24-w4a16-group128", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 2048}, "status": "pending", "targetGpu": "Ascend_910-b3", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.901953+00:00", "framework": "vllm_tokenizer_patch", "intentId": "3e68e483ed784ea797668c6f87c0dbdb", "lastModified": "2026-08-24T19:39:27+00:00", "modelAddress": "https://modelscope.cn/models/neuralmagic/starcoder2-3b-FP8", "repoId": "neuralmagic/starcoder2-3b-FP8", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Ascend_910-b3", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.901996+00:00", "framework": "vllm_tokenizer_patch", "intentId": "d9720955cefa4be99631b865e19632a7", "lastModified": "2026-08-24T18:26:59+00:00", "modelAddress": "https://modelscope.cn/models/nm-testing/nonuniform", "repoId": "nm-testing/nonuniform", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Ascend_910-b3", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.902038+00:00", "framework": "vllm_tokenizer_patch", "intentId": "d9ef0005a582465fb29d4185c0f3b382", "lastModified": "2026-08-24T19:41:12+00:00", "modelAddress": "https://modelscope.cn/models/RedHatAI/starcoder2-15b-quantized.w8a8", "repoId": "RedHatAI/starcoder2-15b-quantized.w8a8", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Ascend_910-b3", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.902078+00:00", "framework": "vllm_tokenizer_patch", "intentId": "8016709760834f878714e8b1261ea15c", "lastModified": "2026-08-24T20:05:18+00:00", "modelAddress": "https://modelscope.cn/models/RedHatAI/Meta-Llama-3.1-8B-Instruct-quantized.w8a16", "repoId": "RedHatAI/Meta-Llama-3.1-8B-Instruct-quantized.w8a16", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Ascend_910-b3", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.902118+00:00", "framework": "vllm_tokenizer_patch", "intentId": "ac6d960d199e45cea54d95d5ff36252a", "lastModified": "2026-08-24T19:59:13+00:00", "modelAddress": "https://modelscope.cn/models/RedHatAI/starcoder2-7b-FP8", "repoId": "RedHatAI/starcoder2-7b-FP8", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Ascend_910-b3", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.902158+00:00", "framework": "vllm_tokenizer_patch", "intentId": "2c0bdf7fa3334d2e81647f4ab3e330be", "lastModified": "2026-08-24T19:38:03+00:00", "modelAddress": "https://modelscope.cn/models/RedHatAI/Phi-3-mini-128k-instruct-quantized.w8a8", "repoId": "RedHatAI/Phi-3-mini-128k-instruct-quantized.w8a8", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Ascend_910-b3", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.902198+00:00", "framework": "vllm_tokenizer_patch", "intentId": "08fa4e65b3e24277bd4e6ece634f1010", "lastModified": "2026-08-24T19:51:52+00:00", "modelAddress": "https://modelscope.cn/models/RedHatAI/Phi-3-small-128k-instruct-quantized.w8a16", "repoId": "RedHatAI/Phi-3-small-128k-instruct-quantized.w8a16", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Ascend_910-b3", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.902238+00:00", "framework": "vllm_tokenizer_patch", "intentId": "f630ae4f9cb548e28cb01339b58ab8d4", "lastModified": "2026-08-24T19:36:07+00:00", "modelAddress": "https://modelscope.cn/models/RedHatAI/Mistral-Nemo-Instruct-2407-quantized.w4a16", "repoId": "RedHatAI/Mistral-Nemo-Instruct-2407-quantized.w4a16", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Ascend_910-b3", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.902279+00:00", "framework": "vllm_tokenizer_patch", "intentId": "f761d106cbaf423bb4308d4afa1b7f0d", "lastModified": "2026-08-24T19:06:55+00:00", "modelAddress": "https://modelscope.cn/models/RedHatAI/gemma-2-2b-it-quantized.w8a8", "repoId": "RedHatAI/gemma-2-2b-it-quantized.w8a8", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Ascend_910-b3", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.902328+00:00", "framework": "vllm_tokenizer_patch", "intentId": "fc336904becf41228759f62f0212c90b", "lastModified": "2026-08-24T20:02:01+00:00", "modelAddress": "https://modelscope.cn/models/RedHatAI/gemma-2-2b-quantized.w8a16", "repoId": "RedHatAI/gemma-2-2b-quantized.w8a16", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Ascend_910-b3", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "1c0a4f69628b2a0ad37a6eee139a4c98afed7d8c93529a9ef42a90756e4e1940", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.902374+00:00", "framework": "vllm_tokenizer_patch", "intentId": "08c7691d32bb4d90b7298dfd819e7bad", "lastModified": "2026-08-24T19:26:17+00:00", "modelAddress": "https://modelscope.cn/models/neuralmagic/gemma-2-2b-it-quantized.w4a16", "repoId": "neuralmagic/gemma-2-2b-it-quantized.w4a16", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Kunlunxin_p-800", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "1c0a4f69628b2a0ad37a6eee139a4c98afed7d8c93529a9ef42a90756e4e1940", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.902428+00:00", "framework": "vllm_tokenizer_patch", "intentId": "efb15b49dc1140cd9fb7b6465d2ae795", "lastModified": "2026-08-26T17:23:58+00:00", "modelAddress": "https://modelscope.cn/models/LiquidAI/LFM2.5-1.2B-Thinking-MLX-8bit", "repoId": "LiquidAI/LFM2.5-1.2B-Thinking-MLX-8bit", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Kunlunxin_p-800", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "1c0a4f69628b2a0ad37a6eee139a4c98afed7d8c93529a9ef42a90756e4e1940", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.902481+00:00", "framework": "vllm_tokenizer_patch", "intentId": "fcac75ee90754e4aabb047c188a5c2bd", "lastModified": "2026-08-24T19:25:45+00:00", "modelAddress": "https://modelscope.cn/models/RedHatAI/Llama-2-7b-chat-quantized.w8a8", "repoId": "RedHatAI/Llama-2-7b-chat-quantized.w8a8", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Kunlunxin_p-800", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "1c0a4f69628b2a0ad37a6eee139a4c98afed7d8c93529a9ef42a90756e4e1940", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.902533+00:00", "framework": "vllm_tokenizer_patch", "intentId": "84ce48f9d3af4311b69254213cbf74f3", "lastModified": "2026-08-24T20:17:00+00:00", "modelAddress": "https://modelscope.cn/models/RedHatAI/Phi-3-mini-128k-instruct-quantized.w8a16", "repoId": "RedHatAI/Phi-3-mini-128k-instruct-quantized.w8a16", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Kunlunxin_p-800", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "1c0a4f69628b2a0ad37a6eee139a4c98afed7d8c93529a9ef42a90756e4e1940", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.902586+00:00", "framework": "vllm_tokenizer_patch", "intentId": "f8d4d6fb6a754946ba496e4c928c2cc6", "lastModified": "2026-08-24T20:08:44+00:00", "modelAddress": "https://modelscope.cn/models/RedHatAI/Phi-3-medium-128k-instruct-quantized.w8a16", "repoId": "RedHatAI/Phi-3-medium-128k-instruct-quantized.w8a16", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Kunlunxin_p-800", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "1c0a4f69628b2a0ad37a6eee139a4c98afed7d8c93529a9ef42a90756e4e1940", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.902638+00:00", "framework": "vllm_tokenizer_patch", "intentId": "f14293c4a0f14d2e95c212a427cacaae", "lastModified": "2026-09-04T12:06:11+00:00", "modelAddress": "https://modelscope.cn/models/sapientinc/HRM-Text-1B", "repoId": "sapientinc/HRM-Text-1B", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Kunlunxin_p-800", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "1c0a4f69628b2a0ad37a6eee139a4c98afed7d8c93529a9ef42a90756e4e1940", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.902696+00:00", "framework": "vllm_tokenizer_patch", "intentId": "ba975573f52349e0852ef3a137a1f5b0", "lastModified": "2026-08-26T16:41:36+00:00", "modelAddress": "https://modelscope.cn/models/aisingapore/Qwen-SEA-LION-v4-32B-IT-8BIT", "repoId": "aisingapore/Qwen-SEA-LION-v4-32B-IT-8BIT", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Kunlunxin_p-800", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "1c0a4f69628b2a0ad37a6eee139a4c98afed7d8c93529a9ef42a90756e4e1940", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.902748+00:00", "framework": "vllm_tokenizer_patch", "intentId": "fe8b827032b44c9a8fca62eaa87771f4", "lastModified": "2026-09-08T15:40:32+00:00", "modelAddress": "https://modelscope.cn/models/JANGQ-AI/Ling-2.6-flash-JANGTQ", "repoId": "JANGQ-AI/Ling-2.6-flash-JANGTQ", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Kunlunxin_p-800", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.902800+00:00", "framework": "vllm", "intentId": "a37a936cb7cd43c0910e93e9a41dcfc2", "lastModified": "2026-08-24T22:55:11+00:00", "modelAddress": "https://modelscope.cn/models/BAAI/AquilaMed-RL", "repoId": "BAAI/AquilaMed-RL", "safeConfigVector": {"dtype": "-", "gpuNum": 1, "maxModelLen": 2048}, "status": "pending", "targetGpu": "Kunlunxin_p-800", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.902862+00:00", "framework": "vllm", "intentId": "5c56b85aa988459d90a0ada7b90f7c65", "lastModified": "2026-08-25T12:06:29+00:00", "modelAddress": "https://modelscope.cn/models/LiquidAI/LFM2.5-8B-A1B", "repoId": "LiquidAI/LFM2.5-8B-A1B", "safeConfigVector": {"dtype": "-", "gpuNum": 1, "maxModelLen": 2048}, "status": "pending", "targetGpu": "Kunlunxin_p-800", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.902923+00:00", "framework": "vllm", "intentId": "6d1ee2a4712944a9af7301c608959c2f", "lastModified": "2026-08-26T20:08:57+00:00", "modelAddress": "https://modelscope.cn/models/LiquidAI/LFM2.5-1.2B-Thinking-MLX-6bit", "repoId": "LiquidAI/LFM2.5-1.2B-Thinking-MLX-6bit", "safeConfigVector": {"dtype": "-", "gpuNum": 1, "maxModelLen": 2048}, "status": "pending", "targetGpu": "Kunlunxin_p-800", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.902984+00:00", "framework": "vllm", "intentId": "6ba55c94b69549f6a46c3e2fd1f30748", "lastModified": "2026-09-04T07:05:43+00:00", "modelAddress": "https://modelscope.cn/models/XHToken/Spark-X2.5-4B-FP8", "repoId": "XHToken/Spark-X2.5-4B-FP8", "safeConfigVector": {"dtype": "-", "gpuNum": 1, "maxModelLen": 2048}, "status": "pending", "targetGpu": "Kunlunxin_p-800", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.903045+00:00", "framework": "vllm", "intentId": "6f987d8a6b684b069404d7739f4e264a", "lastModified": "2026-09-08T13:43:47+00:00", "modelAddress": "https://modelscope.cn/models/JANGQ-AI/Osaurus-AppleScript-8B-JANG_6M", "repoId": "JANGQ-AI/Osaurus-AppleScript-8B-JANG_6M", "safeConfigVector": {"dtype": "-", "gpuNum": 1, "maxModelLen": 2048}, "status": "pending", "targetGpu": "Kunlunxin_p-800", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.903105+00:00", "framework": "vllm", "intentId": "31ef38e789fb4c66b4f04edb84f9e3a2", "lastModified": "2026-08-26T15:57:06+00:00", "modelAddress": "https://modelscope.cn/models/aisingapore/Llama-SEA-LION-v3.5-70B-R-NVFP4", "repoId": "aisingapore/Llama-SEA-LION-v3.5-70B-R-NVFP4", "safeConfigVector": {"dtype": "-", "gpuNum": 1, "maxModelLen": 2048}, "status": "pending", "targetGpu": "Kunlunxin_p-800", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.903166+00:00", "framework": "vllm", "intentId": "2afbe2319ad04b27b0fb69658aec5924", "lastModified": "2026-09-09T06:27:24+00:00", "modelAddress": "https://modelscope.cn/models/ddalcu/Muse-Glimmer-30B-MLX-Serve-4bit", "repoId": "ddalcu/Muse-Glimmer-30B-MLX-Serve-4bit", "safeConfigVector": {"dtype": "-", "gpuNum": 1, "maxModelLen": 2048}, "status": "pending", "targetGpu": "Kunlunxin_p-800", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.903227+00:00", "framework": "vllm", "intentId": "c5a13be12ce64768a6328d1fac3fe4f1", "lastModified": "2026-08-26T18:16:42+00:00", "modelAddress": "https://modelscope.cn/models/aisingapore/SEA-LION-v1-7B-IT-Research", "repoId": "aisingapore/SEA-LION-v1-7B-IT-Research", "safeConfigVector": {"dtype": "-", "gpuNum": 1, "maxModelLen": 2048}, "status": "pending", "targetGpu": "Kunlunxin_p-800", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.903296+00:00", "framework": "vllm_fix_tokenizer", "intentId": "f6df29207c72439bb98584dd3d84ee1b", "lastModified": "2026-08-31T03:35:12+00:00", "modelAddress": "https://modelscope.cn/models/logic65/Qwen3.8-Whittle-w50-18.3B-unrepaired", "repoId": "logic65/Qwen3.8-Whittle-w50-18.3B-unrepaired", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Biren_166m", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.903342+00:00", "framework": "vllm_fix_tokenizer", "intentId": "ef0c7815edd1452691ba64b10cdff3bc", "lastModified": "2026-09-10T14:58:25+00:00", "modelAddress": "https://modelscope.cn/models/webAI-Official/TwIL-LM3", "repoId": "webAI-Official/TwIL-LM3", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Biren_166m", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "0d5bc82787a89d4313672808d1a8629a1fc675c91aa6a08865f770aa24447fbc", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.903388+00:00", "framework": "vllm-patch-tokenizer", "intentId": "de30805fe086458c81ae796d3bc48371", "lastModified": "2026-08-24T19:53:06+00:00", "modelAddress": "https://modelscope.cn/models/neuralmagic/Phi-3-medium-128k-instruct-quantized.w4a16", "repoId": "neuralmagic/Phi-3-medium-128k-instruct-quantized.w4a16", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "hygon_k100-ai", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "0d5bc82787a89d4313672808d1a8629a1fc675c91aa6a08865f770aa24447fbc", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.903433+00:00", "framework": "vllm-patch-tokenizer", "intentId": "95cf443e9cbd48819adf1905e655a1c7", "lastModified": "2026-08-24T20:10:58+00:00", "modelAddress": "https://modelscope.cn/models/neuralmagic/Meta-Llama-3.1-8B-quantized.w8a16", "repoId": "neuralmagic/Meta-Llama-3.1-8B-quantized.w8a16", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "hygon_k100-ai", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "0d5bc82787a89d4313672808d1a8629a1fc675c91aa6a08865f770aa24447fbc", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.903479+00:00", "framework": "vllm-patch-tokenizer", "intentId": "d30f11b86da344449e99ce372525e1e5", "lastModified": "2026-08-24T19:42:52+00:00", "modelAddress": "https://modelscope.cn/models/neuralmagic/starcoder2-7b-FP8", "repoId": "neuralmagic/starcoder2-7b-FP8", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "hygon_k100-ai", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "0d5bc82787a89d4313672808d1a8629a1fc675c91aa6a08865f770aa24447fbc", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.903524+00:00", "framework": "vllm-patch-tokenizer", "intentId": "8bac492945614aae96ea6d66fa777b8a", "lastModified": "2026-08-24T20:21:52+00:00", "modelAddress": "https://modelscope.cn/models/neuralmagic/starcoder2-7b-quantized.w8a16", "repoId": "neuralmagic/starcoder2-7b-quantized.w8a16", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "hygon_k100-ai", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "0d5bc82787a89d4313672808d1a8629a1fc675c91aa6a08865f770aa24447fbc", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.903570+00:00", "framework": "vllm-patch-tokenizer", "intentId": "9d412ad7eeee45bb99d9fb9bb9b7e6dc", "lastModified": "2026-08-24T19:39:20+00:00", "modelAddress": "https://modelscope.cn/models/neuralmagic/starcoder2-3b-quantized.w8a16", "repoId": "neuralmagic/starcoder2-3b-quantized.w8a16", "safeConfigVector": {"gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "hygon_k100-ai", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "ef4c08c913d3f64cae22550f0c4a189795901d4b11e26d8e640e07e857659fb9", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.903616+00:00", "framework": "vllm", "intentId": "498084b8b82a4168bf9ffb6207c75eef", "lastModified": "2026-08-24T19:34:25+00:00", "modelAddress": "https://modelscope.cn/models/neuralmagic/Meta-Llama-3.1-8B-Instruct-FP8", "repoId": "neuralmagic/Meta-Llama-3.1-8B-Instruct-FP8", "safeConfigVector": {"dtype": "-", "gpuNum": 1}, "status": "pending", "targetGpu": "Cambricon_mlu-370-x4", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "ef4c08c913d3f64cae22550f0c4a189795901d4b11e26d8e640e07e857659fb9", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.903671+00:00", "framework": "vllm", "intentId": "f1bb5447bdbc4bdab2e16d516e685ea2", "lastModified": "2026-08-24T20:09:59+00:00", "modelAddress": "https://modelscope.cn/models/neuralmagic/Llama-2-7b-chat-quantized.w8a8", "repoId": "neuralmagic/Llama-2-7b-chat-quantized.w8a8", "safeConfigVector": {"dtype": "-", "gpuNum": 1}, "status": "pending", "targetGpu": "Cambricon_mlu-370-x4", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "ef4c08c913d3f64cae22550f0c4a189795901d4b11e26d8e640e07e857659fb9", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.903721+00:00", "framework": "vllm", "intentId": "f8343194c90045c997c477c5476b98ab", "lastModified": "2026-08-24T20:10:55+00:00", "modelAddress": "https://modelscope.cn/models/RedHatAI/starcoder2-7b-quantized.w8a8", "repoId": "RedHatAI/starcoder2-7b-quantized.w8a8", "safeConfigVector": {"dtype": "-", "gpuNum": 1}, "status": "pending", "targetGpu": "Cambricon_mlu-370-x4", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "ef4c08c913d3f64cae22550f0c4a189795901d4b11e26d8e640e07e857659fb9", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.903771+00:00", "framework": "vllm", "intentId": "8f5f594c28bb4d3fae1c209173bb643b", "lastModified": "2026-08-24T20:02:14+00:00", "modelAddress": "https://modelscope.cn/models/RedHatAI/gemma-2-2b-it-quantized.w8a16", "repoId": "RedHatAI/gemma-2-2b-it-quantized.w8a16", "safeConfigVector": {"dtype": "-", "gpuNum": 1}, "status": "pending", "targetGpu": "Cambricon_mlu-370-x4", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "ef4c08c913d3f64cae22550f0c4a189795901d4b11e26d8e640e07e857659fb9", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.903821+00:00", "framework": "vllm", "intentId": "b306a9e7b2194398af6e47e158694d90", "lastModified": "2026-09-08T15:45:26+00:00", "modelAddress": "https://modelscope.cn/models/JANGQ-AI/AppleScript-8B-JANG_4M", "repoId": "JANGQ-AI/AppleScript-8B-JANG_4M", "safeConfigVector": {"dtype": "-", "gpuNum": 1}, "status": "pending", "targetGpu": "Cambricon_mlu-370-x4", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "f49713b9fe8e3e7683ab2722544c85a44993bcd72b9421c1599f634e68b3a2f7", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.903871+00:00", "framework": "vllm", "intentId": "a8086d26ff2c4556b5e1a7ee2ad4d46e", "lastModified": "2026-08-26T17:29:18+00:00", "modelAddress": "https://modelscope.cn/models/aisingapore/SEA-LION-v1-7B", "repoId": "aisingapore/SEA-LION-v1-7B", "safeConfigVector": {"dtype": "-", "gpuNum": 1}, "status": "pending", "targetGpu": "Cambricon_mlu-370-x4", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.903922+00:00", "framework": "vllm-customized", "intentId": "5824fe118b6e4e42b218b8fe65ffc74f", "lastModified": "2026-09-09T12:19:20+00:00", "modelAddress": "https://modelscope.cn/models/OpenBMB/BitCPM-CANN-8B", "repoId": "OpenBMB/BitCPM-CANN-8B", "safeConfigVector": {"dtype": "-", "gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Cambricon_mlu-370-x8", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.903984+00:00", "framework": "vllm-customized", "intentId": "74631bd7128b4ef6be98a9eda94d4498", "lastModified": "2026-08-24T20:20:16+00:00", "modelAddress": "https://modelscope.cn/models/neuralmagic/starcoder2-15b-FP8", "repoId": "neuralmagic/starcoder2-15b-FP8", "safeConfigVector": {"dtype": "-", "gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Cambricon_mlu-370-x8", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.904045+00:00", "framework": "vllm-customized", "intentId": "8577fa92e3f3400ab164c296e147558b", "lastModified": "2026-08-24T20:01:20+00:00", "modelAddress": "https://modelscope.cn/models/neuralmagic/Llama-3.2-1B-Instruct-quantized.w8a8", "repoId": "neuralmagic/Llama-3.2-1B-Instruct-quantized.w8a8", "safeConfigVector": {"dtype": "-", "gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Cambricon_mlu-370-x8", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.904108+00:00", "framework": "vllm-customized", "intentId": "0b687bc0c3cc492199dc85eeead0d193", "lastModified": "2026-08-24T20:22:00+00:00", "modelAddress": "https://modelscope.cn/models/neuralmagic/gemma-2-2b-it-quantized.w8a16", "repoId": "neuralmagic/gemma-2-2b-it-quantized.w8a16", "safeConfigVector": {"dtype": "-", "gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Cambricon_mlu-370-x8", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.904170+00:00", "framework": "vllm-customized", "intentId": "f0d3b684c28c4ecf9faf0c5605e9c097", "lastModified": "2026-08-26T17:18:42+00:00", "modelAddress": "https://modelscope.cn/models/nv-community/NVIDIA-Nemotron-3-Nano-30B-A3B-NVFP4", "repoId": "nv-community/NVIDIA-Nemotron-3-Nano-30B-A3B-NVFP4", "safeConfigVector": {"dtype": "-", "gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Cambricon_mlu-370-x8", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.904231+00:00", "framework": "vllm-customized", "intentId": "71d638f2dfd247a597073e5d09d58651", "lastModified": "2026-08-24T19:44:39+00:00", "modelAddress": "https://modelscope.cn/models/neuralmagic/gemma-2-2b-it-quantized.w8a8", "repoId": "neuralmagic/gemma-2-2b-it-quantized.w8a8", "safeConfigVector": {"dtype": "-", "gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Cambricon_mlu-370-x8", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.904293+00:00", "framework": "vllm-customized", "intentId": "c81b2f2a5a2043e2b59f2aed2071ad2c", "lastModified": "2026-08-24T20:36:30+00:00", "modelAddress": "https://modelscope.cn/models/neuralmagic/Mistral-Nemo-Instruct-2407-quantized.w4a16", "repoId": "neuralmagic/Mistral-Nemo-Instruct-2407-quantized.w4a16", "safeConfigVector": {"dtype": "-", "gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Cambricon_mlu-370-x8", "taskType": "text-generation"}
{"batchId": "be812550208a4d26813e529259c65486", "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:46:02.904355+00:00", "framework": "vllm-customized", "intentId": "d9c4a210bc964bf99d9bd97b4e8501fd", "lastModified": "2026-08-24T19:40:10+00:00", "modelAddress": "https://modelscope.cn/models/neuralmagic/starcoder2-7b-quantized.w8a8", "repoId": "neuralmagic/starcoder2-7b-quantized.w8a8", "safeConfigVector": {"dtype": "-", "gpuNum": 1, "maxModelLen": 4096}, "status": "pending", "targetGpu": "Cambricon_mlu-370-x8", "taskType": "text-generation"}
{"batchId": "f1d22b7d76084321a5e6a45d08bf0104", "completedAt": "2026-09-21T08:42:21.948173+00:00", "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:36:59.735716+00:00", "framework": "vllm", "intentId": "c2e64998f23a40b4b5d5bdfc07284f0d", "repoId": "aisingapore/Qwen-SEA-LION-v4-32B-IT-4BIT", "safeConfigVector": {"dtype": "-", "gpuNum": 1, "maxModelLen": 2048}, "status": "age_policy_deferred", "targetGpu": "Ascend_910-b4", "taskId": null, "taskType": "text-generation"}
{"batchId": "f1d22b7d76084321a5e6a45d08bf0104", "completedAt": "2026-09-21T08:42:21.948170+00:00", "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:36:59.735654+00:00", "framework": "vllm", "intentId": "bfb716a2925645b59cdac51aed06c793", "repoId": "RedHatAI/gemma-2-2b-it-quantized.w8a16", "safeConfigVector": {"dtype": "-", "gpuNum": 1, "maxModelLen": 2048}, "status": "age_policy_deferred", "targetGpu": "Ascend_910-b4", "taskId": null, "taskType": "text-generation"}
{"batchId": "f1d22b7d76084321a5e6a45d08bf0104", "completedAt": "2026-09-21T08:42:21.948167+00:00", "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configSource": "modelhub_live", "createdAt": "2026-09-21T08:36:59.735597+00:00", "framework": "vllm", "intentId": "b433c64d8fc840f1b2f808089748adc5", "repoId": "RedHatAI/Qwen2-1.5B-Instruct-quantized.w8a8", "safeConfigVector": {"dtype": "-", "gpuNum": 1, "maxModelLen": 2048}, "status": "age_policy_deferred", "targetGpu": "Ascend_910-b4", "taskId": null, "taskType": "text-generation"}

View File

@@ -2,24 +2,24 @@
"agentVersion": "2026.09.20.2",
"checksums": {
".modelhub_state/account_capacity.json": "68d2a8390ad022f2ddcd30841f3860c4535387fa64cb46dc7507eb16837e798f",
".modelhub_state/architecture_compatibility_blacklist.json": "e3137b051a76a893cd9c323d311d8aec5731cfc13d593eb63753065334b00e99",
".modelhub_state/architecture_compatibility_blacklist.json": "c573547154767f9cd6b8b9fc231626a4174e6d7eee45337b9f9557ce48b756c9",
".modelhub_state/architecture_history_backfill.json": "a92348206605da04f75c9e26e7c818b76f259e0b28983f18b6375ef94802f0ab",
".modelhub_state/market_intelligence.json": "6f28c3a75fa495191dcbf043d52f7ea223dca3d5054357ed21abc1d96a8d4eaf",
".modelhub_state/official_capabilities.json": "ba8c394f22e5ab25f87cca4e5d4046438d36ade894da5557c15101ffe0d203a6",
".modelhub_state/outcome_checkpoint.json": "38cf9c229e7987183287b05f5fd6388f23d331dd2c0ca11acc3faebe0eab7c89",
".modelhub_state/market_intelligence.json": "e752c89cd2e65b2bd4e018c92ca2532028af60ca66e17fe5a288375b9be84126",
".modelhub_state/official_capabilities.json": "793f5f74899d2210a0adbe01f638c6ae3c1163a04953c35796f3181b64c913c5",
".modelhub_state/outcome_checkpoint.json": "a295b0568c30ae87c8475f8419f1abc379ff153d5b3227624c10ca7c4e55bac5",
".modelhub_state/queue_cleanup_latest.json": "f4afc5dbbf57b883780778e798518d72ad1770b7d52c2f9efa7ac7bca07813f7",
".modelhub_state/recent_outcomes.jsonl": "192c324383a5677c649ff5959d4c94833cadd97e64612ac0cbb69a2d1c720104",
".modelhub_state/recent_outcomes.jsonl": "29645852c80da99cdb708760921f2785d26a1b7f12d289cc38563a12f682ce62",
".modelhub_state/recovery_active_tasks.jsonl": "8aa976a84cab1f4926684d872f0a65f2378689fdb8714c825b67b1ed965ce0d8",
".modelhub_state/recovery_intents.jsonl": "3eeb5bd84eef7385d7d052e0879a8fefdea7072794e987c17bebb75d0be7c25b",
".modelhub_state/recovery_intents.jsonl": "9464ca60f9d54e4b488cb5a4c7d91d9d201fa7fd635f71feb12abd78acdb074e",
".modelhub_state/routing_intelligence.json": "8f90df7c0bcd028722bfe5ffba7bfcd573a4490cd67df73c92dc262683da0116",
".modelhub_state/submission_exclusions.jsonl": "80315f370e9cc3cdadaf390e12573ef887794214779379c50b7b37b7631e8989",
".modelhub_state/worker_crashes.jsonl": "07505b69e0b050da5f90f8b046a3069d3e1ca2574ca39896bc7d8a66293167f7",
"ledger/submissions.jsonl": "f8823dec161296f53ef93cc1acc6b6ee5f3cc62ad3f2e66b63f14ee7e32d6096",
"outcomes/submissions.jsonl": "065e4ea45771f46633e26e23ae1803e23a704a556ae62a6ff023944f962a7223"
"outcomes/submissions.jsonl": "f88e2a025557533b6598c103f10b9a8840b4314b6e7da49a5707963937574832"
},
"generation": 11343,
"phase": "cycle",
"generation": 11344,
"phase": "intent",
"schemaVersion": 1,
"updatedAt": "2026-09-21T08:43:08.020027+00:00",
"updatedAt": "2026-09-21T08:46:03.051655+00:00",
"writerId": "8b35139af6674067a339a670222d4b67"
}

View File

@@ -137,7 +137,6 @@
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-10T14:29:51.626244+00:00", "modelId": "CohereLabs/tiny-aya-en-thinker", "modelProfile": {"architectures": ["Cohere2ForCausalLM"], "configFingerprint": "71d6e16091ba0289f01acffc8e39bb8c25703488dc49b9e390e08951fe1f2b33", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 6700539432, "estimatedRequiredGiB": 7.522, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "cohere2", "modelscopeFileSize": 6730844445, "modelscopeLicense": "cc-by-nc-4.0", "modelscopeParams": 3350243328, "modelscopeTags": ["license:cc-by-nc-4.0", "model_type:cohere2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:aya", "custom_tag:multilingual", "custom_tag:reasoning", "custom_tag:tiny-aya"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 6730844445}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-10T06:28:56.134725+00:00", "targetGpu": "Iluvatar_mrv-100", "taskId": "4754659", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-10T15:19:40.009053+00:00", "modelId": "primitive-ai/Nex-N2.5-mini-FP8", "modelProfile": {"architectures": ["Qwen3_5MoeForConditionalGeneration"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 38136526544, "estimatedRequiredGiB": 42.65, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_moe", "modelscopeFileSize": 38162746086, "modelscopeLicense": "apache-2.0", "modelscopeParams": 35107181936, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5_moe", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:vllm", "custom_tag:compressed-tensors", "custom_tag:fp8", "custom_tag:quantized", "custom_tag:moe", "custom_tag:nex-n2.5", "custom_tag:agentic", "custom_tag:tool-calling", "custom_tag:single-gpu", "custom_tag:blackwell"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 38162746086}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-10T07:12:26.204766+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4755596", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-patch-tokenizer", "lastSyncTime": "2026-09-10T15:40:28.508808+00:00", "modelId": "CohereLabs/tiny-aya-en-thinker", "modelProfile": {"architectures": ["Cohere2ForCausalLM"], "configFingerprint": "0d5bc82787a89d4313672808d1a8629a1fc675c91aa6a08865f770aa24447fbc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 6700539432, "estimatedRequiredGiB": 7.522, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "cohere2", "modelscopeFileSize": 6730844445, "modelscopeLicense": "cc-by-nc-4.0", "modelscopeParams": 3350243328, "modelscopeTags": ["license:cc-by-nc-4.0", "model_type:cohere2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:aya", "custom_tag:multilingual", "custom_tag:reasoning", "custom_tag:tiny-aya"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 6730844445}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-10T07:33:41.787375+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4756078", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-10T15:56:47.546297+00:00", "modelId": "CohereLabs/tiny-aya-en-thinker", "modelProfile": {"architectures": ["Cohere2ForCausalLM"], "configFingerprint": "a494b1667fab53609b94cc853db1a12c8d3b9fcf98795ecbab3dd5eb3a3c77f6", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 6700539432, "estimatedRequiredGiB": 7.522, "gpuMemoryEvidence": {"memoryGiB": 48.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 48.0, "maximumRepositorySizeGiB": 40.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "cohere2", "modelscopeFileSize": 6730844445, "modelscopeLicense": "cc-by-nc-4.0", "modelscopeParams": 3350243328, "modelscopeTags": ["license:cc-by-nc-4.0", "model_type:cohere2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:aya", "custom_tag:multilingual", "custom_tag:reasoning", "custom_tag:tiny-aya"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 6730844445}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-10T07:50:25.385741+00:00", "targetGpu": "Mthreads_s4000", "taskId": "4756398", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-10T21:33:23.930689+00:00", "modelId": "primitive-ai/Nex-N2.5-mini-FP8", "modelProfile": {"architectures": ["Qwen3_5MoeForConditionalGeneration"], "configFingerprint": "d341b8002cf39d3299fd429479e8c8e954583e49d4fec5a640ddc0f8f0cbd054", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 38136526544, "estimatedRequiredGiB": 42.65, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_moe", "modelscopeFileSize": 38162746086, "modelscopeLicense": "apache-2.0", "modelscopeParams": 35107181936, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5_moe", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:vllm", "custom_tag:compressed-tensors", "custom_tag:fp8", "custom_tag:quantized", "custom_tag:moe", "custom_tag:nex-n2.5", "custom_tag:agentic", "custom_tag:tool-calling", "custom_tag:single-gpu", "custom_tag:blackwell"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 38162746086}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-10T13:27:23.406185+00:00", "targetGpu": "Biren_166m", "taskId": "4760702", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-10T22:47:57.504483+00:00", "modelId": "CohereLabs/tiny-aya-en-thinker", "modelProfile": {"architectures": ["Cohere2ForCausalLM"], "configFingerprint": "1557c77dfe4d3dc1a68ae6b8349e7d094da7eec7752d29442a9d6d27e5f13121", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 6700539432, "estimatedRequiredGiB": 7.522, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "modelhub_preflight_oom:36"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "cohere2", "modelscopeFileSize": 6730844445, "modelscopeLicense": "cc-by-nc-4.0", "modelscopeParams": 3350243328, "modelscopeTags": ["license:cc-by-nc-4.0", "model_type:cohere2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:aya", "custom_tag:multilingual", "custom_tag:reasoning", "custom_tag:tiny-aya"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 6730844445}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-10T14:46:30.893195+00:00", "targetGpu": "Sunrise_pt-200-x1", "taskId": "4761530", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-10T22:47:57.504463+00:00", "modelId": "BAAI/AREX-Turbo", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "1557c77dfe4d3dc1a68ae6b8349e7d094da7eec7752d29442a9d6d27e5f13121", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 9078620368, "estimatedRequiredGiB": 10.18, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "modelhub_preflight_oom:36"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 9109259594, "modelscopeLicense": "apache-2.0", "modelscopeParams": 4539265536, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:agent", "custom_tag:deep-research", "custom_tag:reasoning", "custom_tag:tool-use", "custom_tag:long-context", "custom_tag:qwen3.5", "custom_tag:dense"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 9109259594}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-10T14:46:30.895806+00:00", "targetGpu": "Sunrise_pt-200-x1", "taskId": "4761531", "taskType": "text-generation", "verifyResult": null}
@@ -436,7 +435,6 @@
{"failReason": null, "framework": "vllm-mlu", "lastSyncTime": "2026-09-20T11:10:38.766312+00:00", "modelId": "pfnet/plamo-3-nict-8b-base", "modelProfile": {"architectures": ["Plamo3ForCausalLM"], "configFingerprint": "36ec94e289a8da5fd70b98a2f401fc5159740daf9e2a076da4fd490f34bcbc0b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 16182734928, "estimatedRequiredGiB": 18.09, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "plamo3", "modelscopeFileSize": 16186391993, "modelscopeLicense": "other", "modelscopeParams": 8091348992, "modelscopeTags": ["license:other", "model_type:plamo3", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 16186391993}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:09:17.681306+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "4969311", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-mlu", "lastSyncTime": "2026-09-20T11:10:38.766169+00:00", "modelId": "mlx-community/LFM2.5-1.2B-Thinking-8bit", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "36ec94e289a8da5fd70b98a2f401fc5159740daf9e2a076da4fd490f34bcbc0b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 1243645809, "estimatedRequiredGiB": 1.395, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 1248409935, "modelscopeLicense": "other", "modelscopeParams": 329251584, "modelscopeTags": ["license:other", "model_type:lfm2", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge", "custom_tag:mlx"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 1248409935}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:09:17.694934+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "4969317", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-mlu", "lastSyncTime": "2026-09-20T11:10:38.766358+00:00", "modelId": "inceptionai/Jais-2-8B-Chat", "modelProfile": {"architectures": ["Jais2ForCausalLM"], "configFingerprint": "36ec94e289a8da5fd70b98a2f401fc5159740daf9e2a076da4fd490f34bcbc0b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 16180861064, "estimatedRequiredGiB": 18.097, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "jais2", "modelscopeFileSize": 16193187697, "modelscopeLicense": "apache-2.0", "modelscopeParams": 8090401280, "modelscopeTags": ["license:apache-2.0", "model_type:jais2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 16193187697}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:09:17.840209+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "4969321", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-20T11:10:38.766306+00:00", "modelId": "prithivMLmods/CEERS-2112-14B-Instruct", "modelProfile": {"architectures": ["Qwen2ForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 29540133904, "estimatedRequiredGiB": 33.022, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen2", "modelscopeFileSize": 29547221374, "modelscopeLicense": "apache-2.0", "modelscopeParams": 14770033664, "modelscopeTags": ["license:apache-2.0", "model_type:qwen2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:Code", "custom_tag:Math", "custom_tag:Reasoning", "custom_tag:text-generation-inference", "custom_tag:Reinforcement-learning"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 29547221374}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:09:17.845020+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4969325", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-20T11:10:38.766401+00:00", "modelId": "BAAI/RoboBrain2.5-8B-NV", "modelProfile": {"architectures": ["Qwen3VLForConditionalGeneration"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 17534339552, "estimatedRequiredGiB": 19.609, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_vl", "modelscopeFileSize": 17545918210, "modelscopeLicense": "apache-2.0", "modelscopeParams": 8767123696, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_vl", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 17545918210}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:09:17.837872+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4969323", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T11:10:38.766163+00:00", "modelId": "TokenRhythm/NeoHorse-1-9B", "modelProfile": {"architectures": ["Qwen3_5ForCausalLM"], "configFingerprint": "1c0a4f69628b2a0ad37a6eee139a4c98afed7d8c93529a9ef42a90756e4e1940", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 17907657384, "estimatedRequiredGiB": 20.04, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_text", "modelscopeFileSize": 17931574767, "modelscopeLicense": "Apache License 2.0", "modelscopeParams": 8953803264, "modelscopeTags": ["license:Apache License 2.0", "model_type:qwen3_5_text", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:agentic", "custom_tag:tool-use", "custom_tag:coding", "custom_tag:reasoning", "custom_tag:instruction-following"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 17931574767}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:09:17.942242+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4969330", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T11:10:38.766047+00:00", "modelId": "nv-community/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4", "modelProfile": {"architectures": ["NemotronHForCausalLM"], "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 21561882284, "estimatedRequiredGiB": 24.122, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "nemotron_h", "modelscopeFileSize": 21583784354, "modelscopeLicense": "other", "modelscopeParams": 17820210764, "modelscopeTags": ["license:other", "library:safetensors", "task:text-generation", "custom_tag:nvidia", "custom_tag:pytorch", "custom_tag:nemotron-3.5"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "modelopt", "repositoryOnDiskBytes": 21583784354}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:09:18.141281+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4969338", "taskType": "text-generation", "verifyResult": null}
@@ -552,7 +550,6 @@
{"failReason": null, "framework": "vllm-mlu", "lastSyncTime": "2026-09-20T12:01:00.671231+00:00", "modelId": "neuralmagic/Meta-Llama-3-8B-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "36ec94e289a8da5fd70b98a2f401fc5159740daf9e2a076da4fd490f34bcbc0b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 9084013360, "estimatedRequiredGiB": 10.163, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 9093325202, "modelscopeLicense": "llama3", "modelscopeParams": 8030261248, "modelscopeTags": ["license:llama3", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 9093325202}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:40:37.351352+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "4969964", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-mlu", "lastSyncTime": "2026-09-20T12:01:00.670537+00:00", "modelId": "neuralmagic/Qwen2-7B-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["Qwen2ForCausalLM"], "configFingerprint": "36ec94e289a8da5fd70b98a2f401fc5159740daf9e2a076da4fd490f34bcbc0b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 8708787584, "estimatedRequiredGiB": 9.746, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen2", "modelscopeFileSize": 8720442075, "modelscopeLicense": "apache-2.0", "modelscopeParams": 7615616512, "modelscopeTags": ["license:apache-2.0", "model_type:qwen2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 8720442075}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:40:37.293456+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "4969956", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-20T12:01:00.671152+00:00", "modelId": "LiquidAI/LFM2.5-8B-A1B", "modelProfile": {"architectures": ["Lfm2MoeForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 16936006912, "estimatedRequiredGiB": 18.948, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2_moe", "modelscopeFileSize": 16953948786, "modelscopeLicense": "other", "modelscopeParams": 8467856832, "modelscopeTags": ["license:other", "model_type:lfm2_moe", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 16953948786}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:40:37.353529+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4969966", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-20T12:01:00.671158+00:00", "modelId": "RedHatAI/Meta-Llama-3.1-8B-Instruct-FP8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 9081287016, "estimatedRequiredGiB": 10.159, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 9090504850, "modelscopeLicense": "llama3.1", "modelscopeParams": 8030261248, "modelscopeTags": ["license:llama3.1", "model_type:llama", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:fp8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 9090504850}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:40:37.335603+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4969955", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T12:01:00.670617+00:00", "modelId": "neuralmagic/gemma-2-2b-quantized.w8a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 6746735136, "estimatedRequiredGiB": 7.565, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 6768615496, "modelscopeLicense": "gemma", "modelscopeParams": 3204165888, "modelscopeTags": ["license:gemma", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 6768615496}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:40:37.347716+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4969963", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T12:01:00.670531+00:00", "modelId": "neuralmagic/SmolLM-135M-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "26bfee231695e882b96dee0191bc1dfec25bcad132af96a7ee5f0e26f3cb9fdc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 219850592, "estimatedRequiredGiB": 0.249, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 223236742, "modelscopeLicense": "apache-2.0", "modelscopeParams": 162826560, "modelscopeTags": ["license:apache-2.0", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 223236742}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:40:37.349301+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4969958", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T12:01:00.671359+00:00", "modelId": "RedHatAI/Phi-3-mini-128k-instruct-quantized.w4a16", "modelProfile": {"architectures": ["Phi3ForCausalLM"], "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2263018168, "estimatedRequiredGiB": 2.532, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "phi3", "modelscopeFileSize": 2265562651, "modelscopeLicense": "llama2", "modelscopeParams": 3821079552, "modelscopeTags": ["license:llama2", "model_type:phi3", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 2265562651}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:40:37.355419+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4969959", "taskType": "text-generation", "verifyResult": null}
@@ -666,7 +663,6 @@
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T12:04:34.964288+00:00", "modelId": "neuralmagic/Meta-Llama-3.1-8B-quantized.w8a16", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 9084040952, "estimatedRequiredGiB": 10.163, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 9093251901, "modelscopeLicense": "llama3.1", "modelscopeParams": 8030261248, "modelscopeTags": ["license:llama3.1", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 9093251901}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:02:45.317909+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4970439", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T12:04:34.964242+00:00", "modelId": "neuralmagic/gemma-2-9b-it-quantized.w4a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7963207224, "estimatedRequiredGiB": 8.924, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 7985124714, "modelscopeLicense": "llama2", "modelscopeParams": 10159209984, "modelscopeTags": ["license:llama2", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 7985124714}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:02:45.348395+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4970436", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T12:04:34.964269+00:00", "modelId": "RedHatAI/Llama-3.2-1B-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2024670536, "estimatedRequiredGiB": 2.273, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 2033824831, "modelscopeLicense": "llama3.2", "modelscopeParams": 1498482688, "modelscopeTags": ["license:llama3.2", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:llama", "custom_tag:llama-3", "custom_tag:neuralmagic", "custom_tag:llmcompressor"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 2033824831}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:02:45.354619+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4970444", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T12:04:34.964257+00:00", "modelId": "solidrust/Llama-3-8B-Instruct-v0.4-AWQ", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "6db01b5037d5ed2db74875d2113f8582c97f34bd825637161e54174c9f2cabed", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 5727938576, "estimatedRequiredGiB": 6.412, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 5737300809, "modelscopeLicense": "other", "modelscopeParams": 8030261248, "modelscopeTags": ["license:other", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:4-bit", "custom_tag:AWQ", "custom_tag:text-generation", "custom_tag:autotrain_compatible", "custom_tag:endpoints_compatible", "custom_tag:axolotl", "custom_tag:finetune", "custom_tag:facebook", "custom_tag:meta", "custom_tag:pytorch", "custom_tag:llama", "custom_tag:llama-3"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "awq", "repositoryOnDiskBytes": 5737300809}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:02:45.352471+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4970443", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T12:04:34.964210+00:00", "modelId": "solidrust/Llama-3-11B-Instruct-v0.1-AWQ", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "6db01b5037d5ed2db74875d2113f8582c97f34bd825637161e54174c9f2cabed", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7541230432, "estimatedRequiredGiB": 8.438, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 7550622334, "modelscopeLicense": "other", "modelscopeParams": 11520053248, "modelscopeTags": ["license:other", "model_type:llama", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:4-bit", "custom_tag:AWQ", "custom_tag:text-generation", "custom_tag:autotrain_compatible", "custom_tag:endpoints_compatible", "custom_tag:mergekit", "custom_tag:merge", "custom_tag:facebook", "custom_tag:meta", "custom_tag:pytorch", "custom_tag:llama", "custom_tag:llama-3"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "awq", "repositoryOnDiskBytes": 7550622334}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:02:45.337459+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4970434", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-20T12:21:49.901779+00:00", "modelId": "RedHatAI/starcoder2-3b-FP8", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "cbe5569cdb6c42553436b837dd5b8f0cfa03334642a56f4ff0137f952f8eb84b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 3484485728, "estimatedRequiredGiB": 3.898, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 3487786133, "modelscopeLicense": "other", "modelscopeParams": 3181366272, "modelscopeTags": ["license:other", "model_type:starcoder2", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:fp8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 3487786133}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:09:56.479634+00:00", "targetGpu": "Vastai_va16", "taskId": "4970501", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-20T12:21:49.901761+00:00", "modelId": "mlx-community/gemma-4-e4b-it-OptiQ-4bit", "modelProfile": {"architectures": ["Gemma4ForConditionalGeneration"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7486327132, "estimatedRequiredGiB": 8.403, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma4", "modelscopeFileSize": 7518908360, "modelscopeLicense": "gemma", "modelscopeParams": 2227418442, "modelscopeTags": ["license:gemma", "model_type:gemma4", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:quantized", "custom_tag:mixed-precision", "custom_tag:4bit", "custom_tag:8bit", "custom_tag:optiq", "custom_tag:apple-silicon", "custom_tag:text-generation", "custom_tag:gemma-4"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 7518908360}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:09:56.489765+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4970500", "taskType": "text-generation", "verifyResult": null}
@@ -700,13 +696,11 @@
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T12:21:49.901812+00:00", "modelId": "RedHatAI/starcoder2-15b-FP8", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "ef4c08c913d3f64cae22550f0c4a189795901d4b11e26d8e640e07e857659fb9", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 16564748304, "estimatedRequiredGiB": 18.516, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 16568142065, "modelscopeLicense": "other", "modelscopeParams": 15957889024, "modelscopeTags": ["license:other", "model_type:starcoder2", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:fp8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 16568142065}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:14:41.990095+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "4970591", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T12:21:49.901746+00:00", "modelId": "RedHatAI/SmolLM-135M-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "f49713b9fe8e3e7683ab2722544c85a44993bcd72b9421c1599f634e68b3a2f7", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 219850592, "estimatedRequiredGiB": 0.249, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 223236688, "modelscopeLicense": "apache-2.0", "modelscopeParams": 162826560, "modelscopeTags": ["license:apache-2.0", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 223236688}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:14:42.084586+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "4970594", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-customized", "lastSyncTime": "2026-09-20T12:21:49.902116+00:00", "modelId": "neuralmagic/SmolLM-135M-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "7d3121b7c0c577ab2334a07716695ad3f4016fc14badec71a6e0fbffc883b16d", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 219850592, "estimatedRequiredGiB": 0.249, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 223236742, "modelscopeLicense": "apache-2.0", "modelscopeParams": 162826560, "modelscopeTags": ["license:apache-2.0", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 223236742}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:14:42.070470+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4970598", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-customized", "lastSyncTime": "2026-09-20T12:21:49.901845+00:00", "modelId": "LiquidAI/LFM2.5-1.2B-Thinking-MLX-8bit", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 1243645809, "estimatedRequiredGiB": 1.395, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 1248516007, "modelscopeLicense": "other", "modelscopeParams": 329251584, "modelscopeTags": ["license:other", "model_type:lfm2", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge", "custom_tag:mlx", "custom_tag:reasoning"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 1248516007}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:14:42.077355+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4970592", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-customized", "lastSyncTime": "2026-09-20T12:21:49.901709+00:00", "modelId": "LiquidAI/LFM2.5-1.2B-Thinking-MLX-bf16", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2340697867, "estimatedRequiredGiB": 2.621, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 2345554722, "modelscopeLicense": "other", "modelscopeParams": 1170340608, "modelscopeTags": ["license:other", "model_type:lfm2", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge", "custom_tag:mlx", "custom_tag:reasoning"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 2345554722}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:14:42.081414+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4970595", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T12:21:49.901918+00:00", "modelId": "RedHatAI/gemma-2-27b-it-quantized.w8a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "a494b1667fab53609b94cc853db1a12c8d3b9fcf98795ecbab3dd5eb3a3c77f6", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 30775446656, "estimatedRequiredGiB": 34.419, "gpuMemoryEvidence": {"memoryGiB": 48.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 48.0, "maximumRepositorySizeGiB": 40.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 30797371689, "modelscopeLicense": "gemma", "modelscopeParams": 28406776320, "modelscopeTags": ["license:gemma", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 30797371689}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:14:42.093442+00:00", "targetGpu": "Mthreads_s4000", "taskId": "4970600", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T12:21:49.901767+00:00", "modelId": "neuralmagic/gemma-2-2b-quantized.w8a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 6746735136, "estimatedRequiredGiB": 7.565, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 6768615496, "modelscopeLicense": "gemma", "modelscopeParams": 3204165888, "modelscopeTags": ["license:gemma", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 6768615496}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:14:42.086692+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4970599", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T12:21:49.901615+00:00", "modelId": "neuralmagic/starcoder2-15b-quantized.w8a16", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 17785269848, "estimatedRequiredGiB": 19.88, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 17788663867, "modelscopeLicense": "bigcode-openrail-m", "modelscopeParams": 15957889024, "modelscopeTags": ["license:bigcode-openrail-m", "model_type:starcoder2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 17788663867}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:14:42.090996+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4970597", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-20T12:21:49.901870+00:00", "modelId": "RedHatAI/gemma-4-12B-it-FP8-Dynamic", "modelProfile": {"architectures": ["Gemma4UnifiedForConditionalGeneration"], "configFingerprint": "cbe5569cdb6c42553436b837dd5b8f0cfa03334642a56f4ff0137f952f8eb84b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 15037393456, "estimatedRequiredGiB": 16.842, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma4_unified", "modelscopeFileSize": 15069601989, "modelscopeLicense": "apache-2.0", "modelscopeParams": 12966363184, "modelscopeTags": ["license:apache-2.0", "model_type:gemma4_unified", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "task:image-text-to-text", "custom_tag:fp8", "custom_tag:vllm", "custom_tag:llm-compressor", "custom_tag:compressed-tensors"], "modelscopeTasks": ["text-generation", "image-text-to-text"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 15069601989}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:15:36.946953+00:00", "targetGpu": "Vastai_va16", "taskId": "4970616", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-mlu", "lastSyncTime": "2026-09-20T12:21:49.902150+00:00", "modelId": "mlx-community/LFM2.5-1.2B-Instruct-4bit", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 658540250, "estimatedRequiredGiB": 0.741, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 663396475, "modelscopeLicense": "other", "modelscopeParams": 182975232, "modelscopeTags": ["license:other", "model_type:lfm2", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge", "custom_tag:mlx"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 663396475}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:15:36.970472+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4970618", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T12:21:49.902111+00:00", "modelId": "mlx-community/gemma-4-e2b-it-qat-OptiQ-4bit", "modelProfile": {"architectures": ["Gemma4ForConditionalGeneration"], "configFingerprint": "6db01b5037d5ed2db74875d2113f8582c97f34bd825637161e54174c9f2cabed", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 5222073830, "estimatedRequiredGiB": 5.872, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma4", "modelscopeFileSize": 5254590634, "modelscopeLicense": "apache-2.0", "modelscopeParams": 1140034851, "modelscopeTags": ["license:apache-2.0", "model_type:gemma4", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:quantized", "custom_tag:mixed-precision", "custom_tag:4bit", "custom_tag:8bit", "custom_tag:optiq", "custom_tag:qat", "custom_tag:apple-silicon", "custom_tag:text-generation", "custom_tag:image-text-to-text", "custom_tag:gemma-4"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 5254590634}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:15:37.035881+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4970619", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-mlu", "lastSyncTime": "2026-09-20T12:21:49.901693+00:00", "modelId": "RedHatAI/Llama-3.2-1B-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "36ec94e289a8da5fd70b98a2f401fc5159740daf9e2a076da4fd490f34bcbc0b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2024670536, "estimatedRequiredGiB": 2.273, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 2033824831, "modelscopeLicense": "llama3.2", "modelscopeParams": 1498482688, "modelscopeTags": ["license:llama3.2", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:llama", "custom_tag:llama-3", "custom_tag:neuralmagic", "custom_tag:llmcompressor"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 2033824831}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:19:00.650992+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "4970642", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T12:21:49.901864+00:00", "modelId": "inclusionAI/Ling-3.0-tiny", "modelProfile": {"architectures": ["BailingMoeV3ForCausalLM"], "configFingerprint": "6db01b5037d5ed2db74875d2113f8582c97f34bd825637161e54174c9f2cabed", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 15787992416, "estimatedRequiredGiB": 17.659, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "bailing_hybrid", "modelscopeFileSize": 15801179822, "modelscopeLicense": "mit", "modelscopeParams": 7893392800, "modelscopeTags": ["license:mit", "model_type:bailing_hybrid", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 15801179822}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:19:00.606791+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4970641", "taskType": "text-generation", "verifyResult": null}
@@ -990,13 +984,13 @@
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T08:17:25.672517+00:00", "modelId": "neuralmagic/starcoder2-7b-FP8", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7855165704, "estimatedRequiredGiB": 8.783, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 7858541078, "modelscopeLicense": "other", "modelscopeParams": 7400416256, "modelscopeTags": ["license:other", "model_type:starcoder2", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:fp8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 7858541078}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T00:12:58.738712+00:00", "targetGpu": "Biren_166m", "taskId": "4990140", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T08:22:03.151489+00:00", "modelId": "RedHatAI/starcoder2-3b-quantized.w8a8", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "d341b8002cf39d3299fd429479e8c8e954583e49d4fec5a640ddc0f8f0cbd054", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4093158536, "estimatedRequiredGiB": 4.578, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 4096457676, "modelscopeLicense": "bigcode-openrail-m", "modelscopeParams": 3181366272, "modelscopeTags": ["license:bigcode-openrail-m", "model_type:starcoder2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 4096457676}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T00:17:53.135774+00:00", "targetGpu": "Biren_166m", "taskId": "4990242", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T08:25:31.485626+00:00", "modelId": "aisingapore/Llama-SEA-LION-v3-8B", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "2e9cebc01b20081f0839fa18fe9ca4f7d059509f235699027201fe328177b304", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 16060556376, "estimatedRequiredGiB": 17.964, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 16073634582, "modelscopeLicense": "llama3.1", "modelscopeParams": 8030261248, "modelscopeTags": ["license:llama3.1", "model_type:llama", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 16073634582}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T00:22:06.134784+00:00", "targetGpu": "Sunrise_pt-200-x1", "taskId": "4990307", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": null, "modelId": "RedHatAI/gemma-2-2b-quantized.w8a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 6746735136, "estimatedRequiredGiB": 7.565, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 6768615358, "modelscopeLicense": "gemma", "modelscopeParams": 3204165888, "modelscopeTags": ["license:gemma", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 6768615358}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-21T00:29:04.803696+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4990383", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": null, "modelId": "aisingapore/Gemma-SEA-LION-v4-27B-IT-FP8-Dynamic", "modelProfile": {"architectures": ["Gemma3ForConditionalGeneration"], "configFingerprint": "53aadead5bb69dd6c72976cf6353832080116244f86ffae86297646e5bb48fdc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 29274355224, "estimatedRequiredGiB": 32.761, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma3", "modelscopeFileSize": 29314389092, "modelscopeLicense": "gemma", "modelscopeParams": 27432406640, "modelscopeTags": ["license:gemma", "model_type:gemma3", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 29314389092}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-21T00:29:04.738806+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4990382", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": null, "modelId": "neuralmagic/Llama-3.2-3B-Instruct-FP8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4395007696, "estimatedRequiredGiB": 4.922, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 4404161088, "modelscopeLicense": "llama3.2", "modelscopeParams": 3606752256, "modelscopeTags": ["license:llama3.2", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:llama", "custom_tag:llama-3", "custom_tag:neuralmagic", "custom_tag:llmcompressor"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 4404161088}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-21T00:29:04.722455+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4990380", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": null, "modelId": "XHToken/Spark-X2.5-1.7B", "modelProfile": {"architectures": ["Spark2_5ForCausalLM"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 3415340496, "estimatedRequiredGiB": 3.834, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "spark2_5", "modelscopeFileSize": 3430847031, "modelscopeLicense": "apache-2.0", "modelscopeParams": 1707657216, "modelscopeTags": ["license:apache-2.0", "model_type:spark2_5", "library:safetensors", "library:", "task:text-generation", "custom_tag:llm", "custom_tag:sparkx2_5"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 3430847031}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-21T00:29:04.737218+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4990381", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": null, "modelId": "OpenBMB/MiniCPM5-2B-GPTQ", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2099615880, "estimatedRequiredGiB": 2.358, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 2109841981, "modelscopeLicense": "apache-2.0", "modelscopeParams": 2516756480, "modelscopeTags": ["license:apache-2.0", "model_type:llama", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:minicpm", "custom_tag:minicpm5", "custom_tag:llama", "custom_tag:text-generation", "custom_tag:long-context", "custom_tag:tool-calling", "custom_tag:on-device", "custom_tag:edge-ai"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "awq", "repositoryOnDiskBytes": 2109841981}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-21T00:29:58.598630+00:00", "targetGpu": "Biren_166m", "taskId": "4990404", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-patch-tokenizer", "lastSyncTime": null, "modelId": "neuralmagic/SmolLM-360M-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "4d80215b8854e486ecf078eda8f5db649838852684b9429ab6d7b0d91bb411cf", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 504052632, "estimatedRequiredGiB": 0.567, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 507443106, "modelscopeLicense": "apache-2.0", "modelscopeParams": 409007040, "modelscopeTags": ["license:apache-2.0", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 507443106}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-21T00:36:43.644698+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4990464", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-customized", "lastSyncTime": null, "modelId": "solidrust/Llama-3-8B-Instruct-v0.4-AWQ", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 5727938576, "estimatedRequiredGiB": 6.412, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 5737300809, "modelscopeLicense": "other", "modelscopeParams": 8030261248, "modelscopeTags": ["license:other", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:4-bit", "custom_tag:AWQ", "custom_tag:text-generation", "custom_tag:autotrain_compatible", "custom_tag:endpoints_compatible", "custom_tag:axolotl", "custom_tag:finetune", "custom_tag:facebook", "custom_tag:meta", "custom_tag:pytorch", "custom_tag:llama", "custom_tag:llama-3"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "awq", "repositoryOnDiskBytes": 5737300809}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-21T00:41:04.760036+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4990527", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T08:44:09.587588+00:00", "modelId": "RedHatAI/gemma-2-2b-quantized.w8a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 6746735136, "estimatedRequiredGiB": 7.565, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 6768615358, "modelscopeLicense": "gemma", "modelscopeParams": 3204165888, "modelscopeTags": ["license:gemma", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 6768615358}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T00:29:04.803696+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4990383", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T08:44:09.587545+00:00", "modelId": "aisingapore/Gemma-SEA-LION-v4-27B-IT-FP8-Dynamic", "modelProfile": {"architectures": ["Gemma3ForConditionalGeneration"], "configFingerprint": "53aadead5bb69dd6c72976cf6353832080116244f86ffae86297646e5bb48fdc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 29274355224, "estimatedRequiredGiB": 32.761, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma3", "modelscopeFileSize": 29314389092, "modelscopeLicense": "gemma", "modelscopeParams": 27432406640, "modelscopeTags": ["license:gemma", "model_type:gemma3", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 29314389092}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T00:29:04.738806+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4990382", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T08:44:09.587551+00:00", "modelId": "neuralmagic/Llama-3.2-3B-Instruct-FP8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4395007696, "estimatedRequiredGiB": 4.922, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 4404161088, "modelscopeLicense": "llama3.2", "modelscopeParams": 3606752256, "modelscopeTags": ["license:llama3.2", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:llama", "custom_tag:llama-3", "custom_tag:neuralmagic", "custom_tag:llmcompressor"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 4404161088}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T00:29:04.722455+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4990380", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T08:44:09.587521+00:00", "modelId": "XHToken/Spark-X2.5-1.7B", "modelProfile": {"architectures": ["Spark2_5ForCausalLM"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 3415340496, "estimatedRequiredGiB": 3.834, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "spark2_5", "modelscopeFileSize": 3430847031, "modelscopeLicense": "apache-2.0", "modelscopeParams": 1707657216, "modelscopeTags": ["license:apache-2.0", "model_type:spark2_5", "library:safetensors", "library:", "task:text-generation", "custom_tag:llm", "custom_tag:sparkx2_5"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 3430847031}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T00:29:04.737218+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4990381", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T08:44:09.587576+00:00", "modelId": "OpenBMB/MiniCPM5-2B-GPTQ", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2099615880, "estimatedRequiredGiB": 2.358, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 2109841981, "modelscopeLicense": "apache-2.0", "modelscopeParams": 2516756480, "modelscopeTags": ["license:apache-2.0", "model_type:llama", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:minicpm", "custom_tag:minicpm5", "custom_tag:llama", "custom_tag:text-generation", "custom_tag:long-context", "custom_tag:tool-calling", "custom_tag:on-device", "custom_tag:edge-ai"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "awq", "repositoryOnDiskBytes": 2109841981}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T00:29:58.598630+00:00", "targetGpu": "Biren_166m", "taskId": "4990404", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-patch-tokenizer", "lastSyncTime": "2026-09-21T08:44:09.587484+00:00", "modelId": "neuralmagic/SmolLM-360M-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "4d80215b8854e486ecf078eda8f5db649838852684b9429ab6d7b0d91bb411cf", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 504052632, "estimatedRequiredGiB": 0.567, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 507443106, "modelscopeLicense": "apache-2.0", "modelscopeParams": 409007040, "modelscopeTags": ["license:apache-2.0", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 507443106}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T00:36:43.644698+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4990464", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-customized", "lastSyncTime": "2026-09-21T08:44:09.587538+00:00", "modelId": "solidrust/Llama-3-8B-Instruct-v0.4-AWQ", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 5727938576, "estimatedRequiredGiB": 6.412, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 5737300809, "modelscopeLicense": "other", "modelscopeParams": 8030261248, "modelscopeTags": ["license:other", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:4-bit", "custom_tag:AWQ", "custom_tag:text-generation", "custom_tag:autotrain_compatible", "custom_tag:endpoints_compatible", "custom_tag:axolotl", "custom_tag:finetune", "custom_tag:facebook", "custom_tag:meta", "custom_tag:pytorch", "custom_tag:llama", "custom_tag:llama-3"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "awq", "repositoryOnDiskBytes": 5737300809}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T00:41:04.760036+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4990527", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-mlu", "lastSyncTime": null, "modelId": "neuralmagic/starcoder2-15b-quantized.w8a8", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 17785240496, "estimatedRequiredGiB": 19.88, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 17788612744, "modelscopeLicense": "bigcode-openrail-m", "modelscopeParams": 15957889024, "modelscopeTags": ["license:bigcode-openrail-m", "model_type:starcoder2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 17788612744}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-21T00:51:43.688289+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4990612", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-mlu", "lastSyncTime": null, "modelId": "RedHatAI/starcoder2-15b-FP8", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 16564748304, "estimatedRequiredGiB": 18.516, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 16568142065, "modelscopeLicense": "other", "modelscopeParams": 15957889024, "modelscopeTags": ["license:other", "model_type:starcoder2", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:fp8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 16568142065}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-21T01:10:00.817297+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4990878", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": null, "modelId": "neuralmagic/gemma-2-9b-it-quantized.w8a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 11998647728, "estimatedRequiredGiB": 13.434, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 12020564058, "modelscopeLicense": "gemma", "modelscopeParams": 10159209984, "modelscopeTags": ["license:gemma", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 12020564058}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-21T01:10:00.835785+00:00", "targetGpu": "Biren_166m", "taskId": "4990879", "taskType": "text-generation", "verifyResult": null}