state: generation 11665 (intent)

This commit is contained in:
2026-09-21 19:49:52 +00:00
parent 81301406c2
commit c3c6373460
8 changed files with 367 additions and 389 deletions

View File

@@ -546,7 +546,6 @@
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T12:01:00.671087+00:00", "modelId": "LiquidAI/LFM2.5-2.6B", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "a494b1667fab53609b94cc853db1a12c8d3b9fcf98795ecbab3dd5eb3a3c77f6", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 5394427456, "estimatedRequiredGiB": 6.049, "gpuMemoryEvidence": {"memoryGiB": 48.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 48.0, "maximumRepositorySizeGiB": 40.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 5412393192, "modelscopeLicense": "other", "modelscopeParams": 2697198592, "modelscopeTags": ["license:other", "model_type:lfm2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 5412393192}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:57:37.936950+00:00", "targetGpu": "Mthreads_s4000", "taskId": "4970337", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T12:01:00.671304+00:00", "modelId": "LiquidAI/LFM2.5-1.2B-Instruct-MLX-4bit", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 658540250, "estimatedRequiredGiB": 0.742, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 663548868, "modelscopeLicense": "other", "modelscopeParams": 182975232, "modelscopeTags": ["license:other", "model_type:lfm2", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge", "custom_tag:mlx"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 663548868}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:57:37.944617+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4970338", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T12:01:00.671298+00:00", "modelId": "RedHatAI/Phi-3-mini-128k-instruct-quantized.w8a16", "modelProfile": {"architectures": ["Phi3ForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4020365960, "estimatedRequiredGiB": 4.496, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "phi3", "modelscopeFileSize": 4022810352, "modelscopeLicense": "mit", "modelscopeParams": 3821079552, "modelscopeTags": ["license:mit", "model_type:phi3", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 4022810352}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:58:35.462403+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4970353", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-20T12:01:00.670960+00:00", "modelId": "mlx-community/LFM2.5-1.2B-Instruct-4bit", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 658540250, "estimatedRequiredGiB": 0.741, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 663396475, "modelscopeLicense": "other", "modelscopeParams": 182975232, "modelscopeTags": ["license:other", "model_type:lfm2", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge", "custom_tag:mlx"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 663396475}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:58:35.441724+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4970354", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-20T12:01:00.671026+00:00", "modelId": "Youssofal/Qwen3.5-9B-MTPLX-Optimized-Speed", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 8674988799, "estimatedRequiredGiB": 9.718, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 8695119291, "modelscopeLicense": "apache-2.0", "modelscopeParams": 2415484144, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:apple-silicon", "custom_tag:speculative-decoding", "custom_tag:qwen", "custom_tag:qwen3", "custom_tag:mtp", "custom_tag:mtplx", "custom_tag:local-ai", "custom_tag:qwen3-5", "custom_tag:9b", "custom_tag:6-bit"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 8695119291}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:58:35.439990+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4970355", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T12:01:00.670624+00:00", "modelId": "RedHatAI/SmolLM-135M-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "26bfee231695e882b96dee0191bc1dfec25bcad132af96a7ee5f0e26f3cb9fdc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 219850592, "estimatedRequiredGiB": 0.249, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 223236688, "modelscopeLicense": "apache-2.0", "modelscopeParams": 162826560, "modelscopeTags": ["license:apache-2.0", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 223236688}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:58:35.535522+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4970356", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T12:01:00.671236+00:00", "modelId": "RedHatAI/gemma-2-27b-it-quantized.w8a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 30775446656, "estimatedRequiredGiB": 34.419, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 30797371689, "modelscopeLicense": "gemma", "modelscopeParams": 28406776320, "modelscopeTags": ["license:gemma", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 30797371689}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:58:35.447710+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4970349", "taskType": "text-generation", "verifyResult": null}
@@ -554,7 +553,6 @@
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T12:01:00.671020+00:00", "modelId": "RedHatAI/gemma-4-12B-it-FP8-Dynamic", "modelProfile": {"architectures": ["Gemma4UnifiedForConditionalGeneration"], "configFingerprint": "a494b1667fab53609b94cc853db1a12c8d3b9fcf98795ecbab3dd5eb3a3c77f6", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 15037393456, "estimatedRequiredGiB": 16.842, "gpuMemoryEvidence": {"memoryGiB": 48.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 48.0, "maximumRepositorySizeGiB": 40.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma4_unified", "modelscopeFileSize": 15069601989, "modelscopeLicense": "apache-2.0", "modelscopeParams": 12966363184, "modelscopeTags": ["license:apache-2.0", "model_type:gemma4_unified", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "task:image-text-to-text", "custom_tag:fp8", "custom_tag:vllm", "custom_tag:llm-compressor", "custom_tag:compressed-tensors"], "modelscopeTasks": ["text-generation", "image-text-to-text"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 15069601989}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:58:41.296912+00:00", "targetGpu": "Mthreads_s4000", "taskId": "4970360", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T12:04:34.964276+00:00", "modelId": "neuralmagic/gemma-2-2b-it-quantized.w8a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "a494b1667fab53609b94cc853db1a12c8d3b9fcf98795ecbab3dd5eb3a3c77f6", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4385544296, "estimatedRequiredGiB": 4.926, "gpuMemoryEvidence": {"memoryGiB": 48.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 48.0, "maximumRepositorySizeGiB": 40.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 4407367987, "modelscopeLicense": "gemma", "modelscopeParams": 3204165888, "modelscopeTags": ["license:gemma", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 4407367987}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:01:42.646836+00:00", "targetGpu": "Mthreads_s4000", "taskId": "4970409", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T12:04:34.964288+00:00", "modelId": "neuralmagic/Meta-Llama-3.1-8B-quantized.w8a16", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 9084040952, "estimatedRequiredGiB": 10.163, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 9093251901, "modelscopeLicense": "llama3.1", "modelscopeParams": 8030261248, "modelscopeTags": ["license:llama3.1", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 9093251901}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:02:45.317909+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4970439", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T12:04:34.964242+00:00", "modelId": "neuralmagic/gemma-2-9b-it-quantized.w4a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7963207224, "estimatedRequiredGiB": 8.924, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 7985124714, "modelscopeLicense": "llama2", "modelscopeParams": 10159209984, "modelscopeTags": ["license:llama2", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 7985124714}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:02:45.348395+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4970436", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T12:04:34.964269+00:00", "modelId": "RedHatAI/Llama-3.2-1B-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2024670536, "estimatedRequiredGiB": 2.273, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 2033824831, "modelscopeLicense": "llama3.2", "modelscopeParams": 1498482688, "modelscopeTags": ["license:llama3.2", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:llama", "custom_tag:llama-3", "custom_tag:neuralmagic", "custom_tag:llmcompressor"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 2033824831}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:02:45.354619+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4970444", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-20T12:21:49.901779+00:00", "modelId": "RedHatAI/starcoder2-3b-FP8", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "cbe5569cdb6c42553436b837dd5b8f0cfa03334642a56f4ff0137f952f8eb84b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 3484485728, "estimatedRequiredGiB": 3.898, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 3487786133, "modelscopeLicense": "other", "modelscopeParams": 3181366272, "modelscopeTags": ["license:other", "model_type:starcoder2", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:fp8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 3487786133}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:09:56.479634+00:00", "targetGpu": "Vastai_va16", "taskId": "4970501", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-20T12:21:49.901761+00:00", "modelId": "mlx-community/gemma-4-e4b-it-OptiQ-4bit", "modelProfile": {"architectures": ["Gemma4ForConditionalGeneration"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7486327132, "estimatedRequiredGiB": 8.403, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma4", "modelscopeFileSize": 7518908360, "modelscopeLicense": "gemma", "modelscopeParams": 2227418442, "modelscopeTags": ["license:gemma", "model_type:gemma4", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:quantized", "custom_tag:mixed-precision", "custom_tag:4bit", "custom_tag:8bit", "custom_tag:optiq", "custom_tag:apple-silicon", "custom_tag:text-generation", "custom_tag:gemma-4"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 7518908360}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:09:56.489765+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4970500", "taskType": "text-generation", "verifyResult": null}
@@ -641,7 +639,6 @@
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T18:53:01.473034+00:00", "modelId": "RedHatAI/starcoder2-7b-FP8", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7855165704, "estimatedRequiredGiB": 8.783, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 7858540940, "modelscopeLicense": "other", "modelscopeParams": 7400416256, "modelscopeTags": ["license:other", "model_type:starcoder2", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:fp8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 7858540940}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T10:49:41.611337+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4977491", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T18:53:01.473061+00:00", "modelId": "neuralmagic/gemma-2-9b-it-quantized.w4a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "1c0a4f69628b2a0ad37a6eee139a4c98afed7d8c93529a9ef42a90756e4e1940", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7963207224, "estimatedRequiredGiB": 8.924, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 7985124714, "modelscopeLicense": "llama2", "modelscopeParams": 10159209984, "modelscopeTags": ["license:llama2", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 7985124714}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T10:49:41.634018+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4977490", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T18:53:01.473015+00:00", "modelId": "neuralmagic/Llama-2-7b-chat-quantized.w8a8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7003604832, "estimatedRequiredGiB": 7.829, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 7005626643, "modelscopeLicense": "llama2", "modelscopeParams": 6738415616, "modelscopeTags": ["license:llama2", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 7005626643}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T10:50:01.755721+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4977518", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T18:53:01.472942+00:00", "modelId": "neuralmagic/SmolLM-1.7B-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "6db01b5037d5ed2db74875d2113f8582c97f34bd825637161e54174c9f2cabed", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2014789064, "estimatedRequiredGiB": 2.255, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 2018175215, "modelscopeLicense": "apache-2.0", "modelscopeParams": 1812039680, "modelscopeTags": ["license:apache-2.0", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 2018175215}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T10:51:35.807777+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4977519", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T19:33:34.552434+00:00", "modelId": "neuralmagic/starcoder2-15b-quantized.w8a8", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 17785240496, "estimatedRequiredGiB": 19.88, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 17788612744, "modelscopeLicense": "bigcode-openrail-m", "modelscopeParams": 15957889024, "modelscopeTags": ["license:bigcode-openrail-m", "model_type:starcoder2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 17788612744}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T11:19:51.535102+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4977812", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T19:33:34.552425+00:00", "modelId": "RedHatAI/Phi-3-medium-128k-instruct-quantized.w8a16", "modelProfile": {"architectures": ["Phi3ForCausalLM"], "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 14293356304, "estimatedRequiredGiB": 15.977, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "phi3", "modelscopeFileSize": 14295851911, "modelscopeLicense": "mit", "modelscopeParams": 13960238080, "modelscopeTags": ["license:mit", "model_type:phi3", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 14295851911}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T11:19:51.641136+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4977829", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T19:33:34.552441+00:00", "modelId": "RedHatAI/gemma-2-9b-it-quantized.w4a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7963207224, "estimatedRequiredGiB": 8.924, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 7985124576, "modelscopeLicense": "llama2", "modelscopeParams": 10159209984, "modelscopeTags": ["license:llama2", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 7985124576}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T11:20:03.373586+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4977833", "taskType": "text-generation", "verifyResult": null}
@@ -785,7 +782,6 @@
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T05:17:51.457586+00:00", "modelId": "RedHatAI/starcoder2-15b-quantized.w8a16", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 17785269848, "estimatedRequiredGiB": 19.88, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 17788663591, "modelscopeLicense": "bigcode-openrail-m", "modelscopeParams": 15957889024, "modelscopeTags": ["license:bigcode-openrail-m", "model_type:starcoder2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 17788663591}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T21:15:43.284287+00:00", "targetGpu": "Biren_166m", "taskId": "4987804", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T05:21:25.760194+00:00", "modelId": "aisingapore/Qwen-SEA-LION-v4-32B-IT-8BIT", "modelProfile": {"architectures": ["Qwen3ForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 34327989512, "estimatedRequiredGiB": 38.385, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3", "modelscopeFileSize": 34346039951, "modelscopeLicense": "mit", "modelscopeParams": 32762123264, "modelscopeTags": ["license:mit", "model_type:qwen3", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 34346039951}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T21:17:18.378926+00:00", "targetGpu": "Biren_166m", "taskId": "4987846", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T05:25:05.959941+00:00", "modelId": "neuralmagic/Llama-3.2-3B-Instruct-FP8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "ef4c08c913d3f64cae22550f0c4a189795901d4b11e26d8e640e07e857659fb9", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4395007696, "estimatedRequiredGiB": 4.922, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 4404161088, "modelscopeLicense": "llama3.2", "modelscopeParams": 3606752256, "modelscopeTags": ["license:llama3.2", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:llama", "custom_tag:llama-3", "custom_tag:neuralmagic", "custom_tag:llmcompressor"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 4404161088}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T21:22:47.667525+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "4987928", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-customized", "lastSyncTime": "2026-09-21T05:42:03.443461+00:00", "modelId": "RedHatAI/Meta-Llama-3.1-8B-quantized.w8a16", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 9084040952, "estimatedRequiredGiB": 10.163, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 9093251763, "modelscopeLicense": "llama3.1", "modelscopeParams": 8030261248, "modelscopeTags": ["license:llama3.1", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "recentProfileFeedback": {"attributableFailureCount": 0, "consecutiveFailures": 0, "decisionFailureRate": 0.0, "decisionSuccessRate": 0.0, "decisionTotal": 0, "failureBreakdown": {"ambiguous_runtime": 1}, "failureCount": 1, "failureRate": 1.0, "framework": "vllm-customized", "lastTerminalAt": "2026-09-20T20:49:19.683358+00:00", "modelType": "llama", "pendingCount": 0, "pendingRate": 0.0, "platformFailureCount": 0, "quantizationMethod": "compressed-tensors", "successCount": 0, "successRate": 0.0, "targetGpu": "Cambricon_mlu-370-x8", "taskType": "text-generation", "total": 1, "unresolvedFailureCount": 1}, "repositoryOnDiskBytes": 9093251763}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T21:24:36.675557+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4987943", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T05:42:03.443555+00:00", "modelId": "aisingapore/Gemma-SEA-LION-v3-9B", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "1557c77dfe4d3dc1a68ae6b8349e7d094da7eec7752d29442a9d6d27e5f13121", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 18483466448, "estimatedRequiredGiB": 20.702, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 18523972484, "modelscopeLicense": "gemma", "modelscopeParams": 9241705984, "modelscopeTags": ["license:gemma", "model_type:gemma2", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 18523972484}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T21:26:40.552644+00:00", "targetGpu": "Sunrise_pt-200-x1", "taskId": "4987963", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T05:42:03.443498+00:00", "modelId": "RedHatAI/starcoder2-15b-FP8", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 16564748304, "estimatedRequiredGiB": 18.516, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 16568142065, "modelscopeLicense": "other", "modelscopeParams": 15957889024, "modelscopeTags": ["license:other", "model_type:starcoder2", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:fp8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 16568142065}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T21:26:57.682364+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4987965", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T05:42:03.443590+00:00", "modelId": "mlx-community/Spark-X2.5-4B-OptiQ-4bit", "modelProfile": {"architectures": ["Spark2_5ForCausalLM"], "configFingerprint": "2e9cebc01b20081f0839fa18fe9ca4f7d059509f235699027201fe328177b304", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 3040272206, "estimatedRequiredGiB": 3.409, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "spark2_5", "modelscopeFileSize": 3050595322, "modelscopeLicense": "apache-2.0", "modelscopeParams": 824389120, "modelscopeTags": ["license:apache-2.0", "model_type:spark2_5", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:quantized", "custom_tag:mixed-precision", "custom_tag:4bit", "custom_tag:8bit", "custom_tag:optiq", "custom_tag:apple-silicon", "custom_tag:text-generation", "custom_tag:spark2_5", "custom_tag:long-context"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 3050595322}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T21:28:22.564707+00:00", "targetGpu": "Sunrise_pt-200-x1", "taskId": "4988004", "taskType": "text-generation", "verifyResult": null}
@@ -834,7 +830,6 @@
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T09:32:45.295655+00:00", "modelId": "ewinregirgojr/Qwen3.8-9B-Instruct-Turbo", "modelProfile": {"architectures": ["Qwen3_5ForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 30679155168, "estimatedRequiredGiB": 34.312, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 30702164068, "modelscopeLicense": "apache-2.0", "modelscopeParams": 15180130288, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:qwen", "custom_tag:qwen3", "custom_tag:14b", "custom_tag:deltanet", "custom_tag:linear-attention", "custom_tag:hybrid-attention", "custom_tag:distillation", "custom_tag:pruned", "custom_tag:reasoning", "custom_tag:tool-calling", "custom_tag:agent", "custom_tag:coding", "custom_tag:gguf", "deploy:swingdeploy"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 30702164068}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T01:24:06.954538+00:00", "targetGpu": "Biren_166m", "taskId": "4991108", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T09:32:45.295617+00:00", "modelId": "Youssofal/Qwen3.8-27B-MTPLX-Optimized-Quality-FP16", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 29952487299, "estimatedRequiredGiB": 33.498, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 29973200235, "modelscopeLicense": "apache-2.0", "modelscopeParams": 8027131120, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:apple-silicon", "custom_tag:macos", "custom_tag:m1", "custom_tag:m2", "custom_tag:fp16", "custom_tag:speculative-decoding", "custom_tag:multi-token-prediction", "custom_tag:qwen", "custom_tag:qwen3.8", "custom_tag:mtp", "custom_tag:mtplx", "custom_tag:local-ai", "custom_tag:coding", "custom_tag:qwen3-8", "custom_tag:qwen-3.8", "custom_tag:local-llm", "custom_tag:llm", "custom_tag:m5", "custom_tag:m5-max", "custom_tag:m4", "custom_tag:m3", "custom_tag:macbook-pro", "custom_tag:mac-studio", "custom_tag:opencode", "custom_tag:claude-code", "custom_tag:27b", "custom_tag:qwen3.8-27b", "custom_tag:qwen3-8-27b"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 29973200235}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T01:27:07.435514+00:00", "targetGpu": "Biren_166m", "taskId": "4991159", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T09:54:21.980326+00:00", "modelId": "XHToken/Spark-X2.5-4B", "modelProfile": {"architectures": ["Spark2_5ForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 8224192408, "estimatedRequiredGiB": 9.209, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "spark2_5", "modelscopeFileSize": 8239718268, "modelscopeLicense": "apache-2.0", "modelscopeParams": 4112079360, "modelscopeTags": ["license:apache-2.0", "model_type:spark2_5", "library:safetensors", "library:", "task:text-generation", "custom_tag:llm", "custom_tag:sparkx2_5"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 8239718268}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T01:37:49.775343+00:00", "targetGpu": "Biren_166m", "taskId": "4991287", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-customized", "lastSyncTime": "2026-09-21T09:54:21.980276+00:00", "modelId": "mlx-community/LFM2.5-1.2B-Instruct-4bit", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 658540250, "estimatedRequiredGiB": 0.741, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 663396475, "modelscopeLicense": "other", "modelscopeParams": 182975232, "modelscopeTags": ["license:other", "model_type:lfm2", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge", "custom_tag:mlx"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 663396475}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T01:45:52.835844+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4991428", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-21T09:58:02.958221+00:00", "modelId": "RedHatAI/Phi-3-mini-128k-instruct-FP8", "modelProfile": {"architectures": ["Phi3ForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4018332584, "estimatedRequiredGiB": 4.494, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "phi3", "modelscopeFileSize": 4020783717, "modelscopeLicense": "mit", "modelscopeParams": 3821079552, "modelscopeTags": ["license:mit", "model_type:phi3", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:fp8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "recentProfileFeedback": {"attributableFailureCount": 0, "consecutiveFailures": 0, "decisionFailureRate": 0.0, "decisionSuccessRate": 0.0, "decisionTotal": 0, "failureBreakdown": {"ambiguous_runtime": 1}, "failureCount": 1, "failureRate": 1.0, "framework": "vllm_tokenizer_patch", "lastTerminalAt": "2026-09-21T00:42:10.765015+00:00", "modelType": "phi3", "pendingCount": 0, "pendingRate": 0.0, "platformFailureCount": 0, "quantizationMethod": "compressed-tensors", "successCount": 0, "successRate": 0.0, "targetGpu": "Ascend_910-b4", "taskType": "text-generation", "total": 1, "unresolvedFailureCount": 1}, "repositoryOnDiskBytes": 4020783717}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T01:54:23.239188+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4991516", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T09:58:02.958201+00:00", "modelId": "neuralmagic/Meta-Llama-3.1-8B-Instruct-FP8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 9081287016, "estimatedRequiredGiB": 10.159, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 9090504988, "modelscopeLicense": "llama3.1", "modelscopeParams": 8030261248, "modelscopeTags": ["license:llama3.1", "model_type:llama", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:fp8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 9090504988}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T01:54:23.235347+00:00", "targetGpu": "Biren_166m", "taskId": "4991515", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T10:01:37.498148+00:00", "modelId": "aisingapore/SEA-LION-v1-7B", "modelProfile": {"architectures": ["MPTForCausalLM"], "configFingerprint": "d8df464812ff2fd3c46dc89b0fc6a96370e694fab8e52179f1ada2cdd0ea92b9", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 15003361968, "estimatedRequiredGiB": 16.773, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "mpt", "modelscopeFileSize": 15008115847, "modelscopeLicense": "mit", "modelscopeParams": 7501651968, "modelscopeTags": ["license:mit", "model_type:mpt", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 15008115847}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T01:57:56.851808+00:00", "targetGpu": "Biren_166m", "taskId": "4991645", "taskType": "text-generation", "verifyResult": null}
@@ -922,7 +917,6 @@
{"failReason": null, "framework": "vllm-customized", "lastSyncTime": "2026-09-21T13:58:33.470424+00:00", "modelId": "Arain119/Sophia", "modelProfile": {"architectures": ["SophiaForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2226652224, "estimatedRequiredGiB": 13.067, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "sophia_hybrid", "modelscopeFileSize": 8740644030, "modelscopeLicense": "Apache License 2.0", "modelscopeParams": 1113293776, "modelscopeTags": ["license:Apache License 2.0", "model_type:sophia_hybrid", "library:transformer", "library:safetensors", "library:gguf", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 11692052998}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T05:42:36.743182+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4994779", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-21T14:32:55.759836+00:00", "modelId": "nm-testing/tinyllama-oneshot-w8w8-test-static-shape-change", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 1231270112, "estimatedRequiredGiB": 1.378, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 1233120748, "modelscopeLicense": null, "modelscopeParams": 1100048384, "modelscopeTags": ["model_type:llama", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "recentProfileFeedback": {"attributableFailureCount": 0, "consecutiveFailures": 0, "decisionFailureRate": 0.0, "decisionSuccessRate": 0.0, "decisionTotal": 0, "failureBreakdown": {"ambiguous_runtime": 1}, "failureCount": 1, "failureRate": 1.0, "framework": "vllm_tokenizer_patch", "lastTerminalAt": "2026-09-20T18:18:38.666111+00:00", "modelType": "llama", "pendingCount": 0, "pendingRate": 0.0, "platformFailureCount": 0, "quantizationMethod": "compressed-tensors", "successCount": 0, "successRate": 0.0, "targetGpu": "Ascend_910-b4", "taskType": "text-generation", "total": 1, "unresolvedFailureCount": 1}, "repositoryOnDiskBytes": 1233120748}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T06:25:35.855289+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4995260", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-21T14:32:55.759776+00:00", "modelId": "nm-testing/Meta-Llama-3-8B-Instruct-W8-Channel-A8-Dynamic-Per-Token-Test", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 9084013360, "estimatedRequiredGiB": 10.162, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 9093197700, "modelscopeLicense": null, "modelscopeParams": 8030261248, "modelscopeTags": ["model_type:llama", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "recentProfileFeedback": {"attributableFailureCount": 0, "consecutiveFailures": 0, "decisionFailureRate": 0.0, "decisionSuccessRate": 0.0, "decisionTotal": 0, "failureBreakdown": {"ambiguous_runtime": 1}, "failureCount": 1, "failureRate": 1.0, "framework": "vllm_tokenizer_patch", "lastTerminalAt": "2026-09-20T18:18:38.666111+00:00", "modelType": "llama", "pendingCount": 0, "pendingRate": 0.0, "platformFailureCount": 0, "quantizationMethod": "compressed-tensors", "successCount": 0, "successRate": 0.0, "targetGpu": "Ascend_910-b4", "taskType": "text-generation", "total": 1, "unresolvedFailureCount": 1}, "repositoryOnDiskBytes": 9093197700}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T06:25:35.861357+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4995262", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": null, "modelId": "nm-testing/Meta-Llama-3-8B-Instruct-W8A8-Dyn-Per-Token", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 9084013064, "estimatedRequiredGiB": 10.162, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 9093197403, "modelscopeLicense": null, "modelscopeParams": 8030261248, "modelscopeTags": ["model_type:llama", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "recentProfileFeedback": {"attributableFailureCount": 0, "consecutiveFailures": 0, "decisionFailureRate": 0.0, "decisionSuccessRate": 0.0, "decisionTotal": 0, "failureBreakdown": {"ambiguous_runtime": 1}, "failureCount": 1, "failureRate": 1.0, "framework": "vllm_tokenizer_patch", "lastTerminalAt": "2026-09-20T18:18:38.666111+00:00", "modelType": "llama", "pendingCount": 0, "pendingRate": 0.0, "platformFailureCount": 0, "quantizationMethod": "compressed-tensors", "successCount": 0, "successRate": 0.0, "targetGpu": "Ascend_910-b4", "taskType": "text-generation", "total": 1, "unresolvedFailureCount": 1}, "repositoryOnDiskBytes": 9093197403}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-21T06:25:35.859746+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4995261", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-customized", "lastSyncTime": "2026-09-21T15:02:24.876338+00:00", "modelId": "RedHatAI/starcoder2-15b-quantized.w8a16", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 17785269848, "estimatedRequiredGiB": 19.88, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 17788663591, "modelscopeLicense": "bigcode-openrail-m", "modelscopeParams": 15957889024, "modelscopeTags": ["license:bigcode-openrail-m", "model_type:starcoder2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 17788663591}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T06:33:09.355417+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4995337", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-21T15:02:24.876266+00:00", "modelId": "neuralmagic/gemma-2-2b-it-quantized.w4a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 3403624536, "estimatedRequiredGiB": 3.828, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 3425449487, "modelscopeLicense": "llama2", "modelscopeParams": 3204165888, "modelscopeTags": ["license:llama2", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 3425449487}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T06:49:13.744923+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4995536", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T15:02:24.876381+00:00", "modelId": "primitive-ai/Nemotron-3.5-Lightning-30B-A3B-mixed-INT4-INT8", "modelProfile": {"architectures": ["NemotronHForCausalLM"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 20628596944, "estimatedRequiredGiB": 23.076, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "nemotron_h", "modelscopeFileSize": 20647674585, "modelscopeLicense": "other", "modelscopeParams": 33943909952, "modelscopeTags": ["license:other", "model_type:nemotron_h", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:vllm", "custom_tag:compressed-tensors", "custom_tag:int4", "custom_tag:int8", "custom_tag:mixed-precision", "custom_tag:quantized", "custom_tag:moe", "custom_tag:mamba", "custom_tag:blackwell"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 20647674585}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T06:49:41.154458+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4995549", "taskType": "text-generation", "verifyResult": null}
@@ -1005,12 +999,12 @@
{"failReason": null, "framework": "vllm-patch-tokenizer", "lastSyncTime": "2026-09-21T19:45:40.057919+00:00", "modelId": "neuralmagic/Meta-Llama-3.1-8B-Instruct-quantized.w8a16", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "0d5bc82787a89d4313672808d1a8629a1fc675c91aa6a08865f770aa24447fbc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 9084040952, "estimatedRequiredGiB": 10.163, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 9093261938, "modelscopeLicense": "llama3.1", "modelscopeParams": 8030261248, "modelscopeTags": ["license:llama3.1", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 9093261938}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T11:42:21.093921+00:00", "targetGpu": "hygon_k100-ai", "taskId": "5000138", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-patch-tokenizer", "lastSyncTime": "2026-09-21T19:45:40.057904+00:00", "modelId": "RedHatAI/Llama-3.2-3B-Instruct-FP8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "0d5bc82787a89d4313672808d1a8629a1fc675c91aa6a08865f770aa24447fbc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4395007696, "estimatedRequiredGiB": 4.922, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 4404161034, "modelscopeLicense": "llama3.2", "modelscopeParams": 3606752256, "modelscopeTags": ["license:llama3.2", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:llama", "custom_tag:llama-3", "custom_tag:neuralmagic", "custom_tag:llmcompressor"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 4404161034}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T11:42:38.193150+00:00", "targetGpu": "hygon_k100-ai", "taskId": "5000196", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-customized", "lastSyncTime": "2026-09-21T19:45:40.057852+00:00", "modelId": "OpenBMB/BitCPM-CANN-8B", "modelProfile": {"architectures": ["MiniCPMForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 16370604914, "estimatedRequiredGiB": 18.305, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "minicpm", "modelscopeFileSize": 16378634956, "modelscopeLicense": "apache-2.0", "modelscopeParams": null, "modelscopeTags": ["license:apache-2.0", "model_type:minicpm", "library:pytorch", "library:transformer", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 16378634956}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T11:44:25.650478+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "5000197", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-customized", "lastSyncTime": null, "modelId": "neuralmagic/Llama-3.2-1B-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2024670536, "estimatedRequiredGiB": 2.273, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 2033824885, "modelscopeLicense": "llama3.2", "modelscopeParams": 1498482688, "modelscopeTags": ["license:llama3.2", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:llama", "custom_tag:llama-3", "custom_tag:neuralmagic", "custom_tag:llmcompressor"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "recentProfileFeedback": {"attributableFailureCount": 0, "consecutiveFailures": 0, "decisionFailureRate": 0.0, "decisionSuccessRate": 0.0, "decisionTotal": 0, "failureBreakdown": {"ambiguous_runtime": 2}, "failureCount": 2, "failureRate": 1.0, "framework": "vllm-customized", "lastTerminalAt": "2026-09-21T11:15:42.052853+00:00", "modelType": "llama", "pendingCount": 0, "pendingRate": 0.0, "platformFailureCount": 0, "quantizationMethod": "compressed-tensors", "successCount": 0, "successRate": 0.0, "targetGpu": "Cambricon_mlu-370-x8", "taskType": "text-generation", "total": 2, "unresolvedFailureCount": 2}, "repositoryOnDiskBytes": 2033824885}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-21T11:44:40.201702+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "5000223", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-customized", "lastSyncTime": null, "modelId": "RedHatAI/Llama-3.2-1B-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2024670536, "estimatedRequiredGiB": 2.273, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 2033824831, "modelscopeLicense": "llama3.2", "modelscopeParams": 1498482688, "modelscopeTags": ["license:llama3.2", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:llama", "custom_tag:llama-3", "custom_tag:neuralmagic", "custom_tag:llmcompressor"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "recentProfileFeedback": {"attributableFailureCount": 0, "consecutiveFailures": 0, "decisionFailureRate": 0.0, "decisionSuccessRate": 0.0, "decisionTotal": 0, "failureBreakdown": {"ambiguous_runtime": 2}, "failureCount": 2, "failureRate": 1.0, "framework": "vllm-customized", "lastTerminalAt": "2026-09-21T11:15:42.052853+00:00", "modelType": "llama", "pendingCount": 0, "pendingRate": 0.0, "platformFailureCount": 0, "quantizationMethod": "compressed-tensors", "successCount": 0, "successRate": 0.0, "targetGpu": "Cambricon_mlu-370-x8", "taskType": "text-generation", "total": 2, "unresolvedFailureCount": 2}, "repositoryOnDiskBytes": 2033824831}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-21T11:46:27.062138+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "5000224", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-customized", "lastSyncTime": null, "modelId": "RedHatAI/Phi-3-medium-128k-instruct-quantized.w4a16", "modelProfile": {"architectures": ["Phi3ForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7686303632, "estimatedRequiredGiB": 8.592, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "phi3", "modelscopeFileSize": 7688299781, "modelscopeLicense": "llama2", "modelscopeParams": 13960238080, "modelscopeTags": ["license:llama2", "model_type:phi3", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 7688299781}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-21T11:46:27.192316+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "5000225", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-customized", "lastSyncTime": null, "modelId": "RedHatAI/Qwen2-0.5B-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["Qwen2ForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 903168128, "estimatedRequiredGiB": 1.022, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen2", "modelscopeFileSize": 914658256, "modelscopeLicense": "apache-2.0", "modelscopeParams": 630167424, "modelscopeTags": ["license:apache-2.0", "model_type:qwen2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 914658256}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-21T11:46:27.357049+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "5000226", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-customized", "lastSyncTime": null, "modelId": "RedHatAI/starcoder2-3b-quantized.w8a16", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4093180600, "estimatedRequiredGiB": 4.578, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 4096479058, "modelscopeLicense": "bigcode-openrail-m", "modelscopeParams": 3181366272, "modelscopeTags": ["license:bigcode-openrail-m", "model_type:starcoder2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 4096479058}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-21T11:46:34.674727+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "5000249", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-customized", "lastSyncTime": null, "modelId": "RedHatAI/starcoder2-7b-quantized.w8a16", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7857298896, "estimatedRequiredGiB": 8.785, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 7860673720, "modelscopeLicense": "bigcode-openrail-m", "modelscopeParams": 7400416256, "modelscopeTags": ["license:bigcode-openrail-m", "model_type:starcoder2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 7860673720}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-21T11:48:23.283753+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "5000251", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-customized", "lastSyncTime": "2026-09-21T19:49:37.657088+00:00", "modelId": "neuralmagic/Llama-3.2-1B-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2024670536, "estimatedRequiredGiB": 2.273, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 2033824885, "modelscopeLicense": "llama3.2", "modelscopeParams": 1498482688, "modelscopeTags": ["license:llama3.2", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:llama", "custom_tag:llama-3", "custom_tag:neuralmagic", "custom_tag:llmcompressor"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "recentProfileFeedback": {"attributableFailureCount": 0, "consecutiveFailures": 0, "decisionFailureRate": 0.0, "decisionSuccessRate": 0.0, "decisionTotal": 0, "failureBreakdown": {"ambiguous_runtime": 2}, "failureCount": 2, "failureRate": 1.0, "framework": "vllm-customized", "lastTerminalAt": "2026-09-21T11:15:42.052853+00:00", "modelType": "llama", "pendingCount": 0, "pendingRate": 0.0, "platformFailureCount": 0, "quantizationMethod": "compressed-tensors", "successCount": 0, "successRate": 0.0, "targetGpu": "Cambricon_mlu-370-x8", "taskType": "text-generation", "total": 2, "unresolvedFailureCount": 2}, "repositoryOnDiskBytes": 2033824885}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T11:44:40.201702+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "5000223", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-customized", "lastSyncTime": "2026-09-21T19:49:37.657053+00:00", "modelId": "RedHatAI/Llama-3.2-1B-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2024670536, "estimatedRequiredGiB": 2.273, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 2033824831, "modelscopeLicense": "llama3.2", "modelscopeParams": 1498482688, "modelscopeTags": ["license:llama3.2", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:llama", "custom_tag:llama-3", "custom_tag:neuralmagic", "custom_tag:llmcompressor"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "recentProfileFeedback": {"attributableFailureCount": 0, "consecutiveFailures": 0, "decisionFailureRate": 0.0, "decisionSuccessRate": 0.0, "decisionTotal": 0, "failureBreakdown": {"ambiguous_runtime": 2}, "failureCount": 2, "failureRate": 1.0, "framework": "vllm-customized", "lastTerminalAt": "2026-09-21T11:15:42.052853+00:00", "modelType": "llama", "pendingCount": 0, "pendingRate": 0.0, "platformFailureCount": 0, "quantizationMethod": "compressed-tensors", "successCount": 0, "successRate": 0.0, "targetGpu": "Cambricon_mlu-370-x8", "taskType": "text-generation", "total": 2, "unresolvedFailureCount": 2}, "repositoryOnDiskBytes": 2033824831}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T11:46:27.062138+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "5000224", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-customized", "lastSyncTime": "2026-09-21T19:49:37.657153+00:00", "modelId": "RedHatAI/Phi-3-medium-128k-instruct-quantized.w4a16", "modelProfile": {"architectures": ["Phi3ForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7686303632, "estimatedRequiredGiB": 8.592, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "phi3", "modelscopeFileSize": 7688299781, "modelscopeLicense": "llama2", "modelscopeParams": 13960238080, "modelscopeTags": ["license:llama2", "model_type:phi3", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 7688299781}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T11:46:27.192316+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "5000225", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-customized", "lastSyncTime": "2026-09-21T19:49:37.657126+00:00", "modelId": "RedHatAI/Qwen2-0.5B-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["Qwen2ForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 903168128, "estimatedRequiredGiB": 1.022, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen2", "modelscopeFileSize": 914658256, "modelscopeLicense": "apache-2.0", "modelscopeParams": 630167424, "modelscopeTags": ["license:apache-2.0", "model_type:qwen2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 914658256}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T11:46:27.357049+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "5000226", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-customized", "lastSyncTime": "2026-09-21T19:49:37.657110+00:00", "modelId": "RedHatAI/starcoder2-3b-quantized.w8a16", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4093180600, "estimatedRequiredGiB": 4.578, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 4096479058, "modelscopeLicense": "bigcode-openrail-m", "modelscopeParams": 3181366272, "modelscopeTags": ["license:bigcode-openrail-m", "model_type:starcoder2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 4096479058}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T11:46:34.674727+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "5000249", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-customized", "lastSyncTime": "2026-09-21T19:49:37.657141+00:00", "modelId": "RedHatAI/starcoder2-7b-quantized.w8a16", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7857298896, "estimatedRequiredGiB": 8.785, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 7860673720, "modelscopeLicense": "bigcode-openrail-m", "modelscopeParams": 7400416256, "modelscopeTags": ["license:bigcode-openrail-m", "model_type:starcoder2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 7860673720}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T11:48:23.283753+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "5000251", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": null, "modelId": "neuralmagic/Llama-3.2-3B-Instruct-FP8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4395007696, "estimatedRequiredGiB": 4.922, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 4404161088, "modelscopeLicense": "llama3.2", "modelscopeParams": 3606752256, "modelscopeTags": ["license:llama3.2", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:llama", "custom_tag:llama-3", "custom_tag:neuralmagic", "custom_tag:llmcompressor"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 4404161088}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-21T11:50:46.187626+00:00", "targetGpu": "Ascend_910-b4", "taskId": "5000279", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": null, "modelId": "mlx-community/LFM2.5-1.2B-Instruct-4bit", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "1c0a4f69628b2a0ad37a6eee139a4c98afed7d8c93529a9ef42a90756e4e1940", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 658540250, "estimatedRequiredGiB": 0.741, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 663396475, "modelscopeLicense": "other", "modelscopeParams": 182975232, "modelscopeTags": ["license:other", "model_type:lfm2", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge", "custom_tag:mlx"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 663396475}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-21T11:54:31.274098+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "5000347", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": null, "modelId": "RedHatAI/Phi-3-mini-128k-instruct-FP8", "modelProfile": {"architectures": ["Phi3ForCausalLM"], "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4018332584, "estimatedRequiredGiB": 4.494, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "phi3", "modelscopeFileSize": 4020783717, "modelscopeLicense": "mit", "modelscopeParams": 3821079552, "modelscopeTags": ["license:mit", "model_type:phi3", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:fp8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 4020783717}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-21T11:57:55.153765+00:00", "targetGpu": "Ascend_910-b3", "taskId": "5000419", "taskType": "text-generation", "verifyResult": null}