state: generation 8647 (cycle)
This commit is contained in:
@@ -6,7 +6,6 @@
|
||||
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-04T12:43:05.302380+00:00", "modelId": "solidrust/Llama-3-8B-Instruct-v0.5-AWQ", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "2e9cebc01b20081f0839fa18fe9ca4f7d059509f235699027201fe328177b304", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 5727954960, "estimatedRequiredGiB": 6.412, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "modelhub_preflight_oom:36"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 5737318602, "modelscopeLicense": "other", "modelscopeParams": 8030269440, "modelscopeTags": ["license:other", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:4-bit", "custom_tag:AWQ", "custom_tag:text-generation", "custom_tag:autotrain_compatible", "custom_tag:endpoints_compatible", "custom_tag:axolotl", "custom_tag:finetune", "custom_tag:facebook", "custom_tag:meta", "custom_tag:pytorch", "custom_tag:llama", "custom_tag:llama-3"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "awq", "repositoryOnDiskBytes": 5737318602}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-04T04:34:21.518508+00:00", "targetGpu": "Sunrise_pt-200-x1", "taskId": "4611265", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "llamacpp", "lastSyncTime": "2026-09-04T12:43:05.302313+00:00", "modelId": "b77968543/Spark-X2.5-4B-Q8_0", "modelProfile": {"architectures": [], "configFingerprint": "054a7592edf89d789e18b12765c567c3c34b2ebb58661fdd49255aad03c66d40", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4375021152, "estimatedRequiredGiB": 4.889, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "modelhub_preflight_oom:24"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": null, "modelscopeFileSize": 4375025003, "modelscopeLicense": null, "modelscopeParams": 4112079360, "modelscopeTags": ["library:gguf", "library:", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 4375025003}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-04T04:40:16.372199+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4611339", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-04T15:14:29.722109+00:00", "modelId": "Youssofal/Qwen3.8-27B-MTPLX-Optimized-Quality-FP16", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "d341b8002cf39d3299fd429479e8c8e954583e49d4fec5a640ddc0f8f0cbd054", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 29952487299, "estimatedRequiredGiB": 33.498, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "manufacturer_spec", "sourceUrl": "https://www.birentech.com/news/id6rz98v3obczy77cmxzfgk3/"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 29973198172, "modelscopeLicense": "apache-2.0", "modelscopeParams": 8027131120, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:apple-silicon", "custom_tag:macos", "custom_tag:m1", "custom_tag:m2", "custom_tag:fp16", "custom_tag:speculative-decoding", "custom_tag:multi-token-prediction", "custom_tag:qwen", "custom_tag:qwen3.8", "custom_tag:mtp", "custom_tag:mtplx", "custom_tag:local-ai", "custom_tag:coding"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 29973198172}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-04T07:12:30.054811+00:00", "targetGpu": "Biren_166m", "taskId": "4621809", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm-mlu", "lastSyncTime": "2026-09-04T16:14:29.511220+00:00", "modelId": "Mia-AiLab/Qwen3.8-27B-EXL3-3.5bpw", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "36ec94e289a8da5fd70b98a2f401fc5159740daf9e2a076da4fd490f34bcbc0b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 15338408461, "estimatedRequiredGiB": 17.168, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "modelhub_preflight_oom:49"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 15362064483, "modelscopeLicense": "apache-2.0", "modelscopeParams": 7669052656, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:exl3", "custom_tag:exllamav3", "custom_tag:quantization", "custom_tag:long-context", "custom_tag:speculative-decoding"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "exl3", "repositoryOnDiskBytes": 15362064483}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-04T08:14:12.544960+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "4622655", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-04T20:39:27.718139+00:00", "modelId": "ornith-ai/Ornith-1.5-9B-NVFP4", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 8818103892, "estimatedRequiredGiB": 9.883, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "modelhub_preflight_oom:8"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 8842796317, "modelscopeLicense": "mit", "modelscopeParams": 6728625904, "modelscopeTags": ["license:mit", "model_type:qwen3_5", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "modelopt", "repositoryOnDiskBytes": 8842796317}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-04T12:34:49.403585+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4626360", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-04T22:53:17.234593+00:00", "modelId": "amd/gpt-oss-20b-BF16-w8a8-llmcompressor", "modelProfile": {"architectures": ["GptOssForCausalLM"], "configFingerprint": "2e9cebc01b20081f0839fa18fe9ca4f7d059509f235699027201fe328177b304", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 44193251544, "estimatedRequiredGiB": 49.422, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "modelhub_preflight_oom:36"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gpt_oss", "modelscopeFileSize": 22125413673, "modelscopeLicense": "apache-2.0", "modelscopeParams": 20914757184, "modelscopeTags": ["license:apache-2.0", "model_type:gpt_oss", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:quantized", "custom_tag:int8", "custom_tag:w8a8", "custom_tag:dynamic-quantization", "custom_tag:8-bit", "custom_tag:llm-compressor", "custom_tag:compressed-tensors", "custom_tag:zendnn", "custom_tag:amd", "custom_tag:cpu-inference"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 44221632439}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-04T14:46:47.882785+00:00", "targetGpu": "Sunrise_pt-200-x1", "taskId": "4628625", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-04T23:04:36.601749+00:00", "modelId": "amd/gpt-oss-20b-BF16-w8a8-llmcompressor", "modelProfile": {"architectures": ["GptOssForCausalLM"], "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 44193251544, "estimatedRequiredGiB": 49.422, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "published_card_spec", "sourceUrl": "https://aclanthology.org/2025.emnlp-main.1630.pdf"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gpt_oss", "modelscopeFileSize": 22125413673, "modelscopeLicense": "apache-2.0", "modelscopeParams": 20914757184, "modelscopeTags": ["license:apache-2.0", "model_type:gpt_oss", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:quantized", "custom_tag:int8", "custom_tag:w8a8", "custom_tag:dynamic-quantization", "custom_tag:8-bit", "custom_tag:llm-compressor", "custom_tag:compressed-tensors", "custom_tag:zendnn", "custom_tag:amd", "custom_tag:cpu-inference"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 44221632439}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-04T15:01:53.788146+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4628814", "taskType": "text-generation", "verifyResult": null}
|
||||
@@ -31,7 +30,6 @@
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-05T13:20:02.142845+00:00", "modelId": "r0b0tlab/Qwen3.8-27B-NVFP4-MTP-sm121", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "71d6e16091ba0289f01acffc8e39bb8c25703488dc49b9e390e08951fe1f2b33", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 21921675140, "estimatedRequiredGiB": 24.525, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "modelhub_preflight_oom:50"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 21944922993, "modelscopeLicense": "apache-2.0", "modelscopeParams": 18589348592, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:nvfp4", "custom_tag:vllm", "custom_tag:sm121", "custom_tag:gb10", "custom_tag:dgx-spark", "custom_tag:mtp", "custom_tag:speculative-decoding", "custom_tag:modelopt"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "modelopt", "repositoryOnDiskBytes": 21944922993}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-05T05:17:16.438354+00:00", "targetGpu": "Iluvatar_mrv-100", "taskId": "4641791", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-05T14:51:17.315106+00:00", "modelId": "amd/gpt-oss-20b-BF16-w8a8-llmcompressor", "modelProfile": {"architectures": ["GptOssForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 44193251544, "estimatedRequiredGiB": 49.422, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gpt_oss", "modelscopeFileSize": 44221632439, "modelscopeLicense": "apache-2.0", "modelscopeParams": 20914757184, "modelscopeTags": ["license:apache-2.0", "model_type:gpt_oss", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:quantized", "custom_tag:int8", "custom_tag:w8a8", "custom_tag:dynamic-quantization", "custom_tag:8-bit", "custom_tag:llm-compressor", "custom_tag:compressed-tensors", "custom_tag:zendnn", "custom_tag:amd", "custom_tag:cpu-inference"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 44221632439}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-05T06:48:32.398253+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4643046", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "llamacpp", "lastSyncTime": "2026-09-05T17:06:08.800571+00:00", "modelId": "LiquidAI/LFM2.5-230M-GGUF", "modelProfile": {"architectures": [], "configFingerprint": "c331f3d8906467b7947cec00daebfec4ccad1f351223a1eced4dea452a1012df", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 149080928, "estimatedRequiredGiB": 2.218, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "modelhub_preflight_oom:24"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": null, "modelscopeFileSize": 1984559598, "modelscopeLicense": "other", "modelscopeParams": 229693184, "modelscopeTags": ["license:other", "library:gguf", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:gguf", "custom_tag:llama.cpp", "deploy:swingdeploy"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 1984559598}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-05T09:04:21.884770+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4645136", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-05T17:46:46.041785+00:00", "modelId": "nm-testing/Meta-Llama-3-8B-Instruct-Non-Uniform-compressed-tensors", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "533968aac3b9c90272dfbe2866791f782e077f0dceee5b6ffffce48ba6674a10", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 20041566112, "estimatedRequiredGiB": 22.408, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 20050750527, "modelscopeLicense": null, "modelscopeParams": 8030261248, "modelscopeTags": ["model_type:llama", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "recentProfileFeedback": {"attributableFailureCount": 1, "consecutiveFailures": 1, "decisionFailureRate": 1.0, "decisionSuccessRate": 0.0, "decisionTotal": 1, "failureBreakdown": {"backend_operator": 1}, "failureCount": 1, "failureRate": 1.0, "framework": "vllm", "lastTerminalAt": "2026-09-05T03:28:46.602358+00:00", "modelType": "llama", "pendingCount": 0, "pendingRate": 0.0, "platformFailureCount": 0, "quantizationMethod": "compressed-tensors", "successCount": 0, "successRate": 0.0, "targetGpu": "MetaX_c-500", "taskType": "text-generation", "total": 1, "unresolvedFailureCount": 0}, "repositoryOnDiskBytes": 20050750527}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-05T09:38:43.506267+00:00", "targetGpu": "MetaX_c-500", "taskId": "4645748", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "llamacpp", "lastSyncTime": "2026-09-05T20:12:28.913845+00:00", "modelId": "LiquidAI/LFM2.5-8B-A1B-DSpark-GGUF", "modelProfile": {"architectures": [], "configFingerprint": "e023b149992af42872de3c0c0c07dafd6a98d965e5bdb5a29ec5869e58d45725", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 356491104, "estimatedRequiredGiB": 1.363, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": null, "modelscopeFileSize": 1219879026, "modelscopeLicense": "other", "modelscopeParams": 327707521, "modelscopeTags": ["license:other", "library:gguf", "library:pytorch", "task:text-generation", "custom_tag:speculative-decoding", "custom_tag:dspark", "custom_tag:lfm2", "custom_tag:draft-model", "custom_tag:gguf", "deploy:swingdeploy"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 1219879026}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-05T12:03:37.084154+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4648146", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-05T20:12:28.913926+00:00", "modelId": "siliconflow/Qwen3-Coder-30B-A3B-Instruct-fp8", "modelProfile": {"architectures": ["Qwen3MoeForCausalLM"], "configFingerprint": "533968aac3b9c90272dfbe2866791f782e077f0dceee5b6ffffce48ba6674a10", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 31263441624, "estimatedRequiredGiB": 34.958, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_moe", "modelscopeFileSize": 31280205657, "modelscopeLicense": "Apache License 2.0", "modelscopeParams": 30554505408, "modelscopeTags": ["license:Apache License 2.0", "model_type:qwen3_moe", "library:safetensors", "library:other", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 31280205657}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-05T12:03:37.105603+00:00", "targetGpu": "MetaX_c-500", "taskId": "4648141", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "llamacpp", "lastSyncTime": "2026-09-05T21:15:28.523439+00:00", "modelId": "LiquidAI/LFM2.5-8B-A1B-DSpark-GGUF", "modelProfile": {"architectures": [], "configFingerprint": "0dce7c4be445254429760db90c7969177ecdca460563778d8b3371a37cc1a50e", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 356491104, "estimatedRequiredGiB": 1.363, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": null, "modelscopeFileSize": 1219879026, "modelscopeLicense": "other", "modelscopeParams": 327707521, "modelscopeTags": ["license:other", "library:gguf", "library:pytorch", "task:text-generation", "custom_tag:speculative-decoding", "custom_tag:dspark", "custom_tag:lfm2", "custom_tag:draft-model", "custom_tag:gguf", "deploy:swingdeploy"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 1219879026}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-05T13:07:23.099409+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4649238", "taskType": "text-generation", "verifyResult": null}
|
||||
|
||||
Reference in New Issue
Block a user