|
|
|
|
@@ -175,7 +175,6 @@
|
|
|
|
|
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-06T12:39:38.309054+00:00", "modelId": "groxaxo/Qwen3.8-27B-GPTQ-Pro-4bit-g64-calib128", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "1557c77dfe4d3dc1a68ae6b8349e7d094da7eec7752d29442a9d6d27e5f13121", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 20034587064, "estimatedRequiredGiB": 22.413, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "modelhub_preflight_oom:36"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 20054851725, "modelscopeLicense": "apache-2.0", "modelscopeParams": 27781427952, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:qwen3.8", "custom_tag:qwen3.5-architecture", "custom_tag:gptq-pro", "custom_tag:gptq", "custom_tag:4-bit", "custom_tag:4bit", "custom_tag:mtp", "custom_tag:speculative-decoding", "custom_tag:text-generation", "custom_tag:conversational"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "gptq", "repositoryOnDiskBytes": 20054851725}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-06T04:38:16.005175+00:00", "targetGpu": "Sunrise_pt-200-x1", "taskId": "4661079", "taskType": "text-generation", "verifyResult": null}
|
|
|
|
|
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-06T13:08:06.614699+00:00", "modelId": "RedHatAI/starcoder2-15b-quantized.w8a8", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "533968aac3b9c90272dfbe2866791f782e077f0dceee5b6ffffce48ba6674a10", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 17785240496, "estimatedRequiredGiB": 19.88, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 17788612468, "modelscopeLicense": "bigcode-openrail-m", "modelscopeParams": 15957889024, "modelscopeTags": ["license:bigcode-openrail-m", "model_type:starcoder2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "recentProfileFeedback": {"attributableFailureCount": 2, "consecutiveFailures": 2, "decisionFailureRate": 1.0, "decisionSuccessRate": 0.0, "decisionTotal": 2, "failureBreakdown": {"backend_operator": 2}, "failureCount": 2, "failureRate": 1.0, "framework": "vllm", "lastTerminalAt": "2026-09-05T10:14:38.509276+00:00", "modelType": "starcoder2", "pendingCount": 0, "pendingRate": 0.0, "platformFailureCount": 0, "quantizationMethod": "compressed-tensors", "successCount": 0, "successRate": 0.0, "targetGpu": "MetaX_c-500", "taskType": "text-generation", "total": 2, "unresolvedFailureCount": 0}, "repositoryOnDiskBytes": 17788612468}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-06T05:06:36.456528+00:00", "targetGpu": "MetaX_c-500", "taskId": "4661426", "taskType": "text-generation", "verifyResult": null}
|
|
|
|
|
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-06T13:08:06.614723+00:00", "modelId": "RedHatAI/Phi-3-medium-128k-instruct-FP8", "modelProfile": {"architectures": ["Phi3ForCausalLM"], "configFingerprint": "533968aac3b9c90272dfbe2866791f782e077f0dceee5b6ffffce48ba6674a10", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 14289052464, "estimatedRequiredGiB": 15.972, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "phi3", "modelscopeFileSize": 14291553638, "modelscopeLicense": "mit", "modelscopeParams": 13960238080, "modelscopeTags": ["license:mit", "model_type:phi3", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:fp8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 14291553638}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-06T05:06:36.493823+00:00", "targetGpu": "MetaX_c-500", "taskId": "4661437", "taskType": "text-generation", "verifyResult": null}
|
|
|
|
|
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-06T13:36:29.362958+00:00", "modelId": "RedHatAI/gemma-2-2b-it-quantized.w8a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "533968aac3b9c90272dfbe2866791f782e077f0dceee5b6ffffce48ba6674a10", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4385544296, "estimatedRequiredGiB": 4.926, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 4407367933, "modelscopeLicense": "gemma", "modelscopeParams": 3204165888, "modelscopeTags": ["license:gemma", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 4407367933}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-06T05:34:20.120418+00:00", "targetGpu": "MetaX_c-500", "taskId": "4661809", "taskType": "text-generation", "verifyResult": null}
|
|
|
|
|
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-06T14:42:01.034713+00:00", "modelId": "cyankiwi/Apodex-1.1-mini-AWQ-INT4", "modelProfile": {"architectures": ["Qwen3_5MoeForConditionalGeneration"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 26119812080, "estimatedRequiredGiB": 29.233, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_moe", "modelscopeFileSize": 26156887523, "modelscopeLicense": "apache-2.0", "modelscopeParams": 35951822704, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5_moe", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:agent", "custom_tag:apodex", "custom_tag:arxiv:2608.23283"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 26156887523}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-06T06:30:19.979390+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4662535", "taskType": "text-generation", "verifyResult": null}
|
|
|
|
|
{"failReason": null, "framework": "vllm_0_17_0_corex_4_4_0", "lastSyncTime": "2026-09-06T15:03:37.399221+00:00", "modelId": "cyankiwi/Apodex-1.1-mini-AWQ-INT4", "modelProfile": {"architectures": ["Qwen3_5MoeForConditionalGeneration"], "configFingerprint": "1cc16a7df03cf7b1485e2fda7e5c05e286a546d04e80bdd595c015004b3852a8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 26119812080, "estimatedRequiredGiB": 29.233, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "modelhub_preflight_oom:24"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_moe", "modelscopeFileSize": 26156887523, "modelscopeLicense": "apache-2.0", "modelscopeParams": 35951822704, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5_moe", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:agent", "custom_tag:apodex", "custom_tag:arxiv:2608.23283"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 26156887523}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-06T07:02:35.712857+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4662918", "taskType": "text-generation", "verifyResult": null}
|
|
|
|
|
{"failReason": null, "framework": "vllm-patch-tokenizer", "lastSyncTime": "2026-09-06T15:18:32.207677+00:00", "modelId": "cyankiwi/Apodex-1.1-mini-AWQ-INT4", "modelProfile": {"architectures": ["Qwen3_5MoeForConditionalGeneration"], "configFingerprint": "0d5bc82787a89d4313672808d1a8629a1fc675c91aa6a08865f770aa24447fbc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 26119812080, "estimatedRequiredGiB": 29.233, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_moe", "modelscopeFileSize": 26156887523, "modelscopeLicense": "apache-2.0", "modelscopeParams": 35951822704, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5_moe", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:agent", "custom_tag:apodex", "custom_tag:arxiv:2608.23283"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 26156887523}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-06T07:17:44.185498+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4663094", "taskType": "text-generation", "verifyResult": null}
|
|
|
|
|
@@ -560,8 +559,8 @@
|
|
|
|
|
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-10T14:13:15.207804+00:00", "modelId": "primitive-ai/Nex-N2.5-mini-NVFP4", "modelProfile": {"architectures": ["Qwen3_5MoeForConditionalGeneration"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 23926484304, "estimatedRequiredGiB": 26.777, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_moe", "modelscopeFileSize": 23959625409, "modelscopeLicense": "apache-2.0", "modelscopeParams": 35107212656, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5_moe", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:vllm", "custom_tag:compressed-tensors", "custom_tag:nvfp4", "custom_tag:quantized", "custom_tag:moe", "custom_tag:nex-n2.5", "custom_tag:agentic", "custom_tag:tool-calling", "custom_tag:single-gpu", "custom_tag:blackwell"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 23959625409}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-10T06:12:56.897964+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4754370", "taskType": "text-generation", "verifyResult": null}
|
|
|
|
|
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-10T14:13:15.207812+00:00", "modelId": "primitive-ai/Nex-N2.5-mini-mixed-NVFP4-FP8", "modelProfile": {"architectures": ["Qwen3_5MoeForConditionalGeneration"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 26057989872, "estimatedRequiredGiB": 29.158, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_moe", "modelscopeFileSize": 26090095180, "modelscopeLicense": "apache-2.0", "modelscopeParams": 21416999792, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5_moe", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:vllm", "custom_tag:compressed-tensors", "custom_tag:nvfp4", "custom_tag:fp8", "custom_tag:mixed-precision", "custom_tag:quantized", "custom_tag:moe", "custom_tag:nex-n2.5", "custom_tag:agentic", "custom_tag:tool-calling", "custom_tag:single-gpu", "custom_tag:blackwell"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 26090095180}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-10T06:12:56.900331+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4754369", "taskType": "text-generation", "verifyResult": null}
|
|
|
|
|
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-10T14:13:15.207818+00:00", "modelId": "primitive-ai/Nex-N2.5-mini-FP8", "modelProfile": {"architectures": ["Qwen3_5MoeForConditionalGeneration"], "configFingerprint": "a494b1667fab53609b94cc853db1a12c8d3b9fcf98795ecbab3dd5eb3a3c77f6", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 38136526544, "estimatedRequiredGiB": 42.65, "gpuMemoryEvidence": {"memoryGiB": 48.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 48.0, "maximumRepositorySizeGiB": 40.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_moe", "modelscopeFileSize": 38162746086, "modelscopeLicense": "apache-2.0", "modelscopeParams": 35107181936, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5_moe", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:vllm", "custom_tag:compressed-tensors", "custom_tag:fp8", "custom_tag:quantized", "custom_tag:moe", "custom_tag:nex-n2.5", "custom_tag:agentic", "custom_tag:tool-calling", "custom_tag:single-gpu", "custom_tag:blackwell"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 38162746086}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-10T06:12:56.905628+00:00", "targetGpu": "Mthreads_s4000", "taskId": "4754371", "taskType": "text-generation", "verifyResult": null}
|
|
|
|
|
{"failReason": null, "framework": "llamacpp", "lastSyncTime": null, "modelId": "bartowski/MiniCPM5-2B-GGUF", "modelProfile": {"architectures": [], "configFingerprint": "9ee85a9702ae695242c8b5436b1b856292d4a16c2f839d36bf93fbe98fb66d4e", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 1516997856, "estimatedRequiredGiB": 40.615, "gpuMemoryEvidence": {"memoryGiB": 48.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 48.0, "maximumRepositorySizeGiB": 40.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": null, "modelscopeFileSize": 36341551917, "modelscopeLicense": "apache-2.0", "modelscopeParams": null, "modelscopeTags": ["license:apache-2.0", "library:gguf", "library:pytorch", "task:text-generation", "custom_tag:minicpm", "custom_tag:minicpm5", "custom_tag:llama", "custom_tag:text-generation", "custom_tag:long-context", "custom_tag:tool-calling", "custom_tag:on-device", "custom_tag:edge-ai", "custom_tag:gguf", "deploy:swingdeploy"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 36341551917}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-10T06:28:56.145522+00:00", "targetGpu": "Mthreads_s4000", "taskId": "4754660", "taskType": "text-generation", "verifyResult": null}
|
|
|
|
|
{"failReason": null, "framework": "vllm", "lastSyncTime": null, "modelId": "CohereLabs/tiny-aya-en-thinker", "modelProfile": {"architectures": ["Cohere2ForCausalLM"], "configFingerprint": "71d6e16091ba0289f01acffc8e39bb8c25703488dc49b9e390e08951fe1f2b33", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 6700539432, "estimatedRequiredGiB": 7.522, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "cohere2", "modelscopeFileSize": 6730844445, "modelscopeLicense": "cc-by-nc-4.0", "modelscopeParams": 3350243328, "modelscopeTags": ["license:cc-by-nc-4.0", "model_type:cohere2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:aya", "custom_tag:multilingual", "custom_tag:reasoning", "custom_tag:tiny-aya"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 6730844445}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-10T06:28:56.134725+00:00", "targetGpu": "Iluvatar_mrv-100", "taskId": "4754659", "taskType": "text-generation", "verifyResult": null}
|
|
|
|
|
{"failReason": null, "framework": "llamacpp", "lastSyncTime": "2026-09-10T14:29:51.626225+00:00", "modelId": "bartowski/MiniCPM5-2B-GGUF", "modelProfile": {"architectures": [], "configFingerprint": "9ee85a9702ae695242c8b5436b1b856292d4a16c2f839d36bf93fbe98fb66d4e", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 1516997856, "estimatedRequiredGiB": 40.615, "gpuMemoryEvidence": {"memoryGiB": 48.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 48.0, "maximumRepositorySizeGiB": 40.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": null, "modelscopeFileSize": 36341551917, "modelscopeLicense": "apache-2.0", "modelscopeParams": null, "modelscopeTags": ["license:apache-2.0", "library:gguf", "library:pytorch", "task:text-generation", "custom_tag:minicpm", "custom_tag:minicpm5", "custom_tag:llama", "custom_tag:text-generation", "custom_tag:long-context", "custom_tag:tool-calling", "custom_tag:on-device", "custom_tag:edge-ai", "custom_tag:gguf", "deploy:swingdeploy"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 36341551917}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-10T06:28:56.145522+00:00", "targetGpu": "Mthreads_s4000", "taskId": "4754660", "taskType": "text-generation", "verifyResult": null}
|
|
|
|
|
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-10T14:29:51.626244+00:00", "modelId": "CohereLabs/tiny-aya-en-thinker", "modelProfile": {"architectures": ["Cohere2ForCausalLM"], "configFingerprint": "71d6e16091ba0289f01acffc8e39bb8c25703488dc49b9e390e08951fe1f2b33", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 6700539432, "estimatedRequiredGiB": 7.522, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "cohere2", "modelscopeFileSize": 6730844445, "modelscopeLicense": "cc-by-nc-4.0", "modelscopeParams": 3350243328, "modelscopeTags": ["license:cc-by-nc-4.0", "model_type:cohere2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:aya", "custom_tag:multilingual", "custom_tag:reasoning", "custom_tag:tiny-aya"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 6730844445}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-10T06:28:56.134725+00:00", "targetGpu": "Iluvatar_mrv-100", "taskId": "4754659", "taskType": "text-generation", "verifyResult": null}
|
|
|
|
|
{"failReason": null, "framework": "vllm-patch-tokenizer", "lastSyncTime": null, "modelId": "primitive-ai/Nex-N2.5-mini-NVFP4", "modelProfile": {"architectures": ["Qwen3_5MoeForConditionalGeneration"], "configFingerprint": "0d5bc82787a89d4313672808d1a8629a1fc675c91aa6a08865f770aa24447fbc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 23926484304, "estimatedRequiredGiB": 26.777, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_moe", "modelscopeFileSize": 23959625409, "modelscopeLicense": "apache-2.0", "modelscopeParams": 35107212656, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5_moe", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:vllm", "custom_tag:compressed-tensors", "custom_tag:nvfp4", "custom_tag:quantized", "custom_tag:moe", "custom_tag:nex-n2.5", "custom_tag:agentic", "custom_tag:tool-calling", "custom_tag:single-gpu", "custom_tag:blackwell"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 23959625409}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-10T06:30:53.020882+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4754721", "taskType": "text-generation", "verifyResult": null}
|
|
|
|
|
{"failReason": null, "framework": "vllm-patch-tokenizer", "lastSyncTime": null, "modelId": "primitive-ai/Nex-N2.5-mini-mixed-NVFP4-FP8", "modelProfile": {"architectures": ["Qwen3_5MoeForConditionalGeneration"], "configFingerprint": "0d5bc82787a89d4313672808d1a8629a1fc675c91aa6a08865f770aa24447fbc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 26057989872, "estimatedRequiredGiB": 29.158, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_moe", "modelscopeFileSize": 26090095180, "modelscopeLicense": "apache-2.0", "modelscopeParams": 21416999792, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5_moe", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:vllm", "custom_tag:compressed-tensors", "custom_tag:nvfp4", "custom_tag:fp8", "custom_tag:mixed-precision", "custom_tag:quantized", "custom_tag:moe", "custom_tag:nex-n2.5", "custom_tag:agentic", "custom_tag:tool-calling", "custom_tag:single-gpu", "custom_tag:blackwell"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 26090095180}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-10T06:30:53.025848+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4754722", "taskType": "text-generation", "verifyResult": null}
|
|
|
|
|
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": null, "modelId": "CohereLabs/tiny-aya-en-thinker", "modelProfile": {"architectures": ["Cohere2ForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 6700539432, "estimatedRequiredGiB": 7.522, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "cohere2", "modelscopeFileSize": 6730844445, "modelscopeLicense": "cc-by-nc-4.0", "modelscopeParams": 3350243328, "modelscopeTags": ["license:cc-by-nc-4.0", "model_type:cohere2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:aya", "custom_tag:multilingual", "custom_tag:reasoning", "custom_tag:tiny-aya"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 6730844445}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-10T06:38:16.883366+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4754820", "taskType": "text-generation", "verifyResult": null}
|
|
|
|
|
|