state: generation 11676 (intent)
This commit is contained in:
@@ -469,7 +469,6 @@
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T12:01:00.670706+00:00", "modelId": "RedHatAI/starcoder2-15b-quantized.w8a16", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "ef4c08c913d3f64cae22550f0c4a189795901d4b11e26d8e640e07e857659fb9", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 17785269848, "estimatedRequiredGiB": 19.88, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 17788663591, "modelscopeLicense": "bigcode-openrail-m", "modelscopeParams": 15957889024, "modelscopeTags": ["license:bigcode-openrail-m", "model_type:starcoder2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 17788663591}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:40:44.501444+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "4969969", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T12:01:00.671081+00:00", "modelId": "RedHatAI/SmolLM-360M-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "f49713b9fe8e3e7683ab2722544c85a44993bcd72b9421c1599f634e68b3a2f7", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 504052632, "estimatedRequiredGiB": 0.567, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 507443052, "modelscopeLicense": "apache-2.0", "modelscopeParams": 409007040, "modelscopeTags": ["license:apache-2.0", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 507443052}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:40:44.505215+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "4969967", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T12:01:00.671015+00:00", "modelId": "LiquidAI/LFM2.5-2.6B", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "ef4c08c913d3f64cae22550f0c4a189795901d4b11e26d8e640e07e857659fb9", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 5394427456, "estimatedRequiredGiB": 6.049, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 5412393192, "modelscopeLicense": "other", "modelscopeParams": 2697198592, "modelscopeTags": ["license:other", "model_type:lfm2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 5412393192}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:40:44.507984+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "4969968", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T12:01:00.670523+00:00", "modelId": "neuralmagic/Qwen2-0.5B-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["Qwen2ForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 903168128, "estimatedRequiredGiB": 1.022, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen2", "modelscopeFileSize": 914658310, "modelscopeLicense": "apache-2.0", "modelscopeParams": 630167424, "modelscopeTags": ["license:apache-2.0", "model_type:qwen2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 914658310}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:41:42.237083+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4969989", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-20T12:01:00.671348+00:00", "modelId": "LiquidAI/LFM2.5-1.2B-Thinking-MLX-8bit", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "cbe5569cdb6c42553436b837dd5b8f0cfa03334642a56f4ff0137f952f8eb84b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 1243645809, "estimatedRequiredGiB": 1.395, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 1248516007, "modelscopeLicense": "other", "modelscopeParams": 329251584, "modelscopeTags": ["license:other", "model_type:lfm2", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge", "custom_tag:mlx", "custom_tag:reasoning"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 1248516007}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:41:42.249155+00:00", "targetGpu": "Vastai_va16", "taskId": "4969994", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-20T12:01:00.670941+00:00", "modelId": "LiquidAI/LFM2.5-1.2B-Thinking-MLX-bf16", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "cbe5569cdb6c42553436b837dd5b8f0cfa03334642a56f4ff0137f952f8eb84b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2340697867, "estimatedRequiredGiB": 2.621, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 2345554722, "modelscopeLicense": "other", "modelscopeParams": 1170340608, "modelscopeTags": ["license:other", "model_type:lfm2", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge", "custom_tag:mlx", "custom_tag:reasoning"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 2345554722}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:41:42.236412+00:00", "targetGpu": "Vastai_va16", "taskId": "4969990", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-20T12:01:00.671225+00:00", "modelId": "LiquidAI/LFM2.5-1.2B-Thinking-MLX-6bit", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "cbe5569cdb6c42553436b837dd5b8f0cfa03334642a56f4ff0137f952f8eb84b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 951093016, "estimatedRequiredGiB": 1.068, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 955963132, "modelscopeLicense": "other", "modelscopeParams": 256113408, "modelscopeTags": ["license:other", "model_type:lfm2", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge", "custom_tag:mlx", "custom_tag:reasoning"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 955963132}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:41:42.238130+00:00", "targetGpu": "Vastai_va16", "taskId": "4969985", "taskType": "text-generation", "verifyResult": null}
|
||||
@@ -548,7 +547,6 @@
|
||||
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-20T12:01:00.671026+00:00", "modelId": "Youssofal/Qwen3.5-9B-MTPLX-Optimized-Speed", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 8674988799, "estimatedRequiredGiB": 9.718, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 8695119291, "modelscopeLicense": "apache-2.0", "modelscopeParams": 2415484144, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:apple-silicon", "custom_tag:speculative-decoding", "custom_tag:qwen", "custom_tag:qwen3", "custom_tag:mtp", "custom_tag:mtplx", "custom_tag:local-ai", "custom_tag:qwen3-5", "custom_tag:9b", "custom_tag:6-bit"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 8695119291}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:58:35.439990+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4970355", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T12:01:00.670624+00:00", "modelId": "RedHatAI/SmolLM-135M-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "26bfee231695e882b96dee0191bc1dfec25bcad132af96a7ee5f0e26f3cb9fdc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 219850592, "estimatedRequiredGiB": 0.249, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 223236688, "modelscopeLicense": "apache-2.0", "modelscopeParams": 162826560, "modelscopeTags": ["license:apache-2.0", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 223236688}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:58:35.535522+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4970356", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T12:01:00.671236+00:00", "modelId": "RedHatAI/gemma-2-27b-it-quantized.w8a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 30775446656, "estimatedRequiredGiB": 34.419, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 30797371689, "modelscopeLicense": "gemma", "modelscopeParams": 28406776320, "modelscopeTags": ["license:gemma", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 30797371689}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:58:35.447710+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4970349", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T12:01:00.670884+00:00", "modelId": "BAAI/RoboBrain2.5-8B-MT", "modelProfile": {"architectures": ["Qwen3VLForConditionalGeneration"], "configFingerprint": "6db01b5037d5ed2db74875d2113f8582c97f34bd825637161e54174c9f2cabed", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 17534339552, "estimatedRequiredGiB": 19.609, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_vl", "modelscopeFileSize": 17545918174, "modelscopeLicense": "apache-2.0", "modelscopeParams": 8767123696, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_vl", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 17545918174}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:58:35.455197+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4970358", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T12:01:00.671020+00:00", "modelId": "RedHatAI/gemma-4-12B-it-FP8-Dynamic", "modelProfile": {"architectures": ["Gemma4UnifiedForConditionalGeneration"], "configFingerprint": "a494b1667fab53609b94cc853db1a12c8d3b9fcf98795ecbab3dd5eb3a3c77f6", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 15037393456, "estimatedRequiredGiB": 16.842, "gpuMemoryEvidence": {"memoryGiB": 48.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 48.0, "maximumRepositorySizeGiB": 40.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma4_unified", "modelscopeFileSize": 15069601989, "modelscopeLicense": "apache-2.0", "modelscopeParams": 12966363184, "modelscopeTags": ["license:apache-2.0", "model_type:gemma4_unified", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "task:image-text-to-text", "custom_tag:fp8", "custom_tag:vllm", "custom_tag:llm-compressor", "custom_tag:compressed-tensors"], "modelscopeTasks": ["text-generation", "image-text-to-text"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 15069601989}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T03:58:41.296912+00:00", "targetGpu": "Mthreads_s4000", "taskId": "4970360", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T12:04:34.964276+00:00", "modelId": "neuralmagic/gemma-2-2b-it-quantized.w8a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "a494b1667fab53609b94cc853db1a12c8d3b9fcf98795ecbab3dd5eb3a3c77f6", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4385544296, "estimatedRequiredGiB": 4.926, "gpuMemoryEvidence": {"memoryGiB": 48.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 48.0, "maximumRepositorySizeGiB": 40.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 4407367987, "modelscopeLicense": "gemma", "modelscopeParams": 3204165888, "modelscopeTags": ["license:gemma", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 4407367987}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:01:42.646836+00:00", "targetGpu": "Mthreads_s4000", "taskId": "4970409", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T12:04:34.964288+00:00", "modelId": "neuralmagic/Meta-Llama-3.1-8B-quantized.w8a16", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 9084040952, "estimatedRequiredGiB": 10.163, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 9093251901, "modelscopeLicense": "llama3.1", "modelscopeParams": 8030261248, "modelscopeTags": ["license:llama3.1", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 9093251901}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:02:45.317909+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4970439", "taskType": "text-generation", "verifyResult": null}
|
||||
@@ -593,7 +591,6 @@
|
||||
{"failReason": null, "framework": "vllm-mlu", "lastSyncTime": "2026-09-20T12:34:29.856347+00:00", "modelId": "RedHatAI/gemma-4-12B-it-FP8-Dynamic", "modelProfile": {"architectures": ["Gemma4UnifiedForConditionalGeneration"], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 15037393456, "estimatedRequiredGiB": 16.842, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma4_unified", "modelscopeFileSize": 15069601989, "modelscopeLicense": "apache-2.0", "modelscopeParams": 12966363184, "modelscopeTags": ["license:apache-2.0", "model_type:gemma4_unified", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "task:image-text-to-text", "custom_tag:fp8", "custom_tag:vllm", "custom_tag:llm-compressor", "custom_tag:compressed-tensors"], "modelscopeTasks": ["text-generation", "image-text-to-text"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 15069601989}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:31:05.659043+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4970789", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T12:46:14.462961+00:00", "modelId": "RedHatAI/gemma-2-2b-it-quantized.w4a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 3403624536, "estimatedRequiredGiB": 3.828, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 3425449433, "modelscopeLicense": "llama2", "modelscopeParams": 3204165888, "modelscopeTags": ["license:llama2", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 3425449433}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:36:48.898217+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4970869", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-20T12:46:14.462938+00:00", "modelId": "empero-ai/Qwen3.8-35B-A3B-Distill", "modelProfile": {"architectures": ["Qwen3_5MoeForConditionalGeneration"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 71903778536, "estimatedRequiredGiB": 80.392, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_moe", "modelscopeFileSize": 71933964484, "modelscopeLicense": "apache-2.0", "modelscopeParams": 35951822704, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5_moe", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:empero-ai", "custom_tag:qwen3.6", "custom_tag:qwen3.8", "custom_tag:distillation", "custom_tag:reasoning", "custom_tag:function-calling", "custom_tag:moe", "custom_tag:sft"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "recentProfileFeedback": {"attributableFailureCount": 0, "consecutiveFailures": 0, "decisionFailureRate": 0.0, "decisionSuccessRate": 0.0, "decisionTotal": 0, "failureBreakdown": {"ambiguous_runtime": 2}, "failureCount": 2, "failureRate": 1.0, "framework": "vllm_fix_tokenizer", "lastTerminalAt": "2026-09-15T22:59:41.110897+00:00", "modelType": "qwen3_5_moe", "pendingCount": 0, "pendingRate": 0.0, "platformFailureCount": 0, "quantizationMethod": "none", "successCount": 0, "successRate": 0.0, "targetGpu": "Kunlunxin_p-800", "taskType": "text-generation", "total": 2, "unresolvedFailureCount": 2}, "repositoryOnDiskBytes": 71933964484}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:42:19.067096+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4970926", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm-mlu", "lastSyncTime": "2026-09-20T12:46:14.462954+00:00", "modelId": "mlx-community/LFM2.5-1.2B-Thinking-6bit", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 951093016, "estimatedRequiredGiB": 1.068, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 955857175, "modelscopeLicense": "other", "modelscopeParams": 256113408, "modelscopeTags": ["license:other", "model_type:lfm2", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge", "custom_tag:mlx"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 955857175}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:44:53.386157+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4970949", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm_0_17_0_corex_4_4_0", "lastSyncTime": "2026-09-20T12:46:14.462918+00:00", "modelId": "mlx-community/LFM2.5-1.2B-Thinking-8bit", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "1cc16a7df03cf7b1485e2fda7e5c05e286a546d04e80bdd595c015004b3852a8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 1243645809, "estimatedRequiredGiB": 1.395, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 1248409935, "modelscopeLicense": "other", "modelscopeParams": 329251584, "modelscopeTags": ["license:other", "model_type:lfm2", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge", "custom_tag:mlx"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 1248409935}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T04:44:53.390325+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4970950", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T13:05:06.219610+00:00", "modelId": "RedHatAI/gemma-4-12B-it-FP8-Dynamic", "modelProfile": {"architectures": ["Gemma4UnifiedForConditionalGeneration"], "configFingerprint": "d341b8002cf39d3299fd429479e8c8e954583e49d4fec5a640ddc0f8f0cbd054", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 15037393456, "estimatedRequiredGiB": 16.842, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma4_unified", "modelscopeFileSize": 15069601989, "modelscopeLicense": "apache-2.0", "modelscopeParams": 12966363184, "modelscopeTags": ["license:apache-2.0", "model_type:gemma4_unified", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "task:image-text-to-text", "custom_tag:fp8", "custom_tag:vllm", "custom_tag:llm-compressor", "custom_tag:compressed-tensors"], "modelscopeTasks": ["text-generation", "image-text-to-text"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 15069601989}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T05:01:56.033898+00:00", "targetGpu": "Biren_166m", "taskId": "4971137", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm-patch-tokenizer", "lastSyncTime": "2026-09-20T13:18:18.977292+00:00", "modelId": "RedHatAI/gemma-4-12B-it-FP8-Dynamic", "modelProfile": {"architectures": ["Gemma4UnifiedForConditionalGeneration"], "configFingerprint": "0d5bc82787a89d4313672808d1a8629a1fc675c91aa6a08865f770aa24447fbc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 15037393456, "estimatedRequiredGiB": 16.842, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma4_unified", "modelscopeFileSize": 15069601989, "modelscopeLicense": "apache-2.0", "modelscopeParams": 12966363184, "modelscopeTags": ["license:apache-2.0", "model_type:gemma4_unified", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "task:image-text-to-text", "custom_tag:fp8", "custom_tag:vllm", "custom_tag:llm-compressor", "custom_tag:compressed-tensors"], "modelscopeTasks": ["text-generation", "image-text-to-text"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 15069601989}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T05:16:13.324611+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4971311", "taskType": "text-generation", "verifyResult": null}
|
||||
@@ -1005,7 +1002,7 @@
|
||||
{"failReason": null, "framework": "vllm-customized", "lastSyncTime": "2026-09-21T19:49:37.657110+00:00", "modelId": "RedHatAI/starcoder2-3b-quantized.w8a16", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4093180600, "estimatedRequiredGiB": 4.578, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 4096479058, "modelscopeLicense": "bigcode-openrail-m", "modelscopeParams": 3181366272, "modelscopeTags": ["license:bigcode-openrail-m", "model_type:starcoder2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 4096479058}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T11:46:34.674727+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "5000249", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm-customized", "lastSyncTime": "2026-09-21T19:49:37.657141+00:00", "modelId": "RedHatAI/starcoder2-7b-quantized.w8a16", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7857298896, "estimatedRequiredGiB": 8.785, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 7860673720, "modelscopeLicense": "bigcode-openrail-m", "modelscopeParams": 7400416256, "modelscopeTags": ["license:bigcode-openrail-m", "model_type:starcoder2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 7860673720}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T11:48:23.283753+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "5000251", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T19:52:34.157576+00:00", "modelId": "neuralmagic/Llama-3.2-3B-Instruct-FP8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4395007696, "estimatedRequiredGiB": 4.922, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 4404161088, "modelscopeLicense": "llama3.2", "modelscopeParams": 3606752256, "modelscopeTags": ["license:llama3.2", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:llama", "custom_tag:llama-3", "custom_tag:neuralmagic", "custom_tag:llmcompressor"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 4404161088}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T11:50:46.187626+00:00", "targetGpu": "Ascend_910-b4", "taskId": "5000279", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": null, "modelId": "mlx-community/LFM2.5-1.2B-Instruct-4bit", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "1c0a4f69628b2a0ad37a6eee139a4c98afed7d8c93529a9ef42a90756e4e1940", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 658540250, "estimatedRequiredGiB": 0.741, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 663396475, "modelscopeLicense": "other", "modelscopeParams": 182975232, "modelscopeTags": ["license:other", "model_type:lfm2", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge", "custom_tag:mlx"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 663396475}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-21T11:54:31.274098+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "5000347", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-21T19:57:11.769093+00:00", "modelId": "mlx-community/LFM2.5-1.2B-Instruct-4bit", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "1c0a4f69628b2a0ad37a6eee139a4c98afed7d8c93529a9ef42a90756e4e1940", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 658540250, "estimatedRequiredGiB": 0.741, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 663396475, "modelscopeLicense": "other", "modelscopeParams": 182975232, "modelscopeTags": ["license:other", "model_type:lfm2", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge", "custom_tag:mlx"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 663396475}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T11:54:31.274098+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "5000347", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": null, "modelId": "RedHatAI/Phi-3-mini-128k-instruct-FP8", "modelProfile": {"architectures": ["Phi3ForCausalLM"], "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4018332584, "estimatedRequiredGiB": 4.494, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "phi3", "modelscopeFileSize": 4020783717, "modelscopeLicense": "mit", "modelscopeParams": 3821079552, "modelscopeTags": ["license:mit", "model_type:phi3", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:fp8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 4020783717}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-21T11:57:55.153765+00:00", "targetGpu": "Ascend_910-b3", "taskId": "5000419", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": null, "modelId": "aisingapore/Gemma-SEA-LION-v4-27B-IT-FP8-Dynamic", "modelProfile": {"architectures": ["Gemma3ForConditionalGeneration"], "configFingerprint": "1c0a4f69628b2a0ad37a6eee139a4c98afed7d8c93529a9ef42a90756e4e1940", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 29274355224, "estimatedRequiredGiB": 32.761, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma3", "modelscopeFileSize": 29314389092, "modelscopeLicense": "gemma", "modelscopeParams": 27432406640, "modelscopeTags": ["license:gemma", "model_type:gemma3", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 29314389092}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-21T11:57:55.143712+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "5000420", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": null, "modelId": "RedHatAI/Phi-3-medium-128k-instruct-quantized.w4a16", "modelProfile": {"architectures": ["Phi3ForCausalLM"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7686303632, "estimatedRequiredGiB": 8.592, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "phi3", "modelscopeFileSize": 7688299781, "modelscopeLicense": "llama2", "modelscopeParams": 13960238080, "modelscopeTags": ["license:llama2", "model_type:phi3", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 7688299781}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-21T11:57:55.174530+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "5000425", "taskType": "text-generation", "verifyResult": null}
|
||||
|
||||
Reference in New Issue
Block a user