state: generation 10848 (cycle)

This commit is contained in:
2026-09-20 19:53:11 +00:00
parent 8d32fad412
commit 963f424e94
8 changed files with 1814 additions and 1818 deletions

View File

@@ -2146,7 +2146,7 @@
"taskType": "text-generation" "taskType": "text-generation"
} }
}, },
"generatedAt": "2026-09-20T19:49:27.061617+00:00", "generatedAt": "2026-09-20T19:53:02.825174+00:00",
"summary": { "summary": {
"activeBlockCount": 105, "activeBlockCount": 105,
"byGpuFramework": { "byGpuFramework": {

View File

@@ -425,7 +425,7 @@
} }
}, },
"frameworkUpdatedAt": null, "frameworkUpdatedAt": null,
"generatedAt": "2026-09-20T19:52:01.341682+00:00", "generatedAt": "2026-09-20T19:53:10.852703+00:00",
"gpuStats": { "gpuStats": {
"Ascend_910-b3": { "Ascend_910-b3": {
"available": true, "available": true,

View File

@@ -1,5 +1,5 @@
{ {
"catalogUpdatedAt": "2026-09-20T19:52:01.341682+00:00", "catalogUpdatedAt": "2026-09-20T19:53:10.852703+00:00",
"configuredTaskTypes": [ "configuredTaskTypes": [
"text-generation" "text-generation"
], ],
@@ -56,7 +56,7 @@
"time-series-forecasting" "time-series-forecasting"
], ],
"errors": [], "errors": [],
"generatedAt": "2026-09-20T19:52:01.341682+00:00", "generatedAt": "2026-09-20T19:53:10.852703+00:00",
"gpuCatalog": { "gpuCatalog": {
"Ascend_910-b3": { "Ascend_910-b3": {
"canVerify": true, "canVerify": true,
@@ -6383,6 +6383,6 @@
"updateTime": "2025-12-22 08:59:53" "updateTime": "2025-12-22 08:59:53"
} }
], ],
"taskTreeUpdatedAt": "2026-09-20T19:52:01.341682+00:00", "taskTreeUpdatedAt": "2026-09-20T19:53:10.852703+00:00",
"version": 1 "version": 1
} }

File diff suppressed because it is too large Load Diff

File diff suppressed because it is too large Load Diff

View File

@@ -418,7 +418,6 @@
{"framework": "vllm-mlu", "modelAddress": "https://modelscope.cn/models/mlx-community/Spark-X2.5-4B-OptiQ-4bit", "modelId": "mlx-community/Spark-X2.5-4B-OptiQ-4bit", "submitTime": "2026-09-19T17:54:56.831790+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4961250", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-mlu-cambricon-mlu-370-x8"} {"framework": "vllm-mlu", "modelAddress": "https://modelscope.cn/models/mlx-community/Spark-X2.5-4B-OptiQ-4bit", "modelId": "mlx-community/Spark-X2.5-4B-OptiQ-4bit", "submitTime": "2026-09-19T17:54:56.831790+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4961250", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-mlu-cambricon-mlu-370-x8"}
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/mlx-community/Spark-X2.5-4B-OptiQ-4bit", "modelId": "mlx-community/Spark-X2.5-4B-OptiQ-4bit", "submitTime": "2026-09-19T18:15:01.430761+00:00", "targetGpu": "Biren_166m", "taskId": "4961485", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-biren-166m"} {"framework": "vllm", "modelAddress": "https://modelscope.cn/models/mlx-community/Spark-X2.5-4B-OptiQ-4bit", "modelId": "mlx-community/Spark-X2.5-4B-OptiQ-4bit", "submitTime": "2026-09-19T18:15:01.430761+00:00", "targetGpu": "Biren_166m", "taskId": "4961485", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-biren-166m"}
{"framework": "llamacpp", "modelAddress": "https://modelscope.cn/models/chenqi2026/7a-8elite", "modelId": "chenqi2026/7a-8elite", "submitTime": "2026-09-20T02:52:43.741402+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4968996", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-llamacpp-hygon-k100-ai"} {"framework": "llamacpp", "modelAddress": "https://modelscope.cn/models/chenqi2026/7a-8elite", "modelId": "chenqi2026/7a-8elite", "submitTime": "2026-09-20T02:52:43.741402+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4968996", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-llamacpp-hygon-k100-ai"}
{"framework": "vllm_tokenizer_patch", "modelAddress": "https://modelscope.cn/models/nv-community/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4", "modelId": "nv-community/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4", "submitTime": "2026-09-20T02:52:43.745571+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4969000", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-tokenizer-patch-ascend-910-b4"}
{"framework": "vllm_tokenizer_patch", "modelAddress": "https://modelscope.cn/models/RedHatAI/gemma-4-12B-it-FP8-Dynamic", "modelId": "RedHatAI/gemma-4-12B-it-FP8-Dynamic", "submitTime": "2026-09-20T02:52:43.738452+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4968995", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-tokenizer-patch-ascend-910-b4"} {"framework": "vllm_tokenizer_patch", "modelAddress": "https://modelscope.cn/models/RedHatAI/gemma-4-12B-it-FP8-Dynamic", "modelId": "RedHatAI/gemma-4-12B-it-FP8-Dynamic", "submitTime": "2026-09-20T02:52:43.738452+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4968995", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-tokenizer-patch-ascend-910-b4"}
{"framework": "vllm_tokenizer_patch", "modelAddress": "https://modelscope.cn/models/mlx-community/LFM2.5-1.2B-Thinking-6bit", "modelId": "mlx-community/LFM2.5-1.2B-Thinking-6bit", "submitTime": "2026-09-20T02:52:43.835992+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4969002", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-tokenizer-patch-ascend-910-b4"} {"framework": "vllm_tokenizer_patch", "modelAddress": "https://modelscope.cn/models/mlx-community/LFM2.5-1.2B-Thinking-6bit", "modelId": "mlx-community/LFM2.5-1.2B-Thinking-6bit", "submitTime": "2026-09-20T02:52:43.835992+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4969002", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-tokenizer-patch-ascend-910-b4"}
{"framework": "vllm_tokenizer_patch", "modelAddress": "https://modelscope.cn/models/Youssofal/Qwen3.8-27B-MTPLX-Bare-Speed", "modelId": "Youssofal/Qwen3.8-27B-MTPLX-Bare-Speed", "submitTime": "2026-09-20T02:52:43.750661+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4969005", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-tokenizer-patch-ascend-910-b4"} {"framework": "vllm_tokenizer_patch", "modelAddress": "https://modelscope.cn/models/Youssofal/Qwen3.8-27B-MTPLX-Bare-Speed", "modelId": "Youssofal/Qwen3.8-27B-MTPLX-Bare-Speed", "submitTime": "2026-09-20T02:52:43.750661+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4969005", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-tokenizer-patch-ascend-910-b4"}
@@ -488,7 +487,6 @@
{"framework": "vllm_fix_tokenizer", "modelAddress": "https://modelscope.cn/models/prithivMLmods/CEERS-2112-14B-Instruct", "modelId": "prithivMLmods/CEERS-2112-14B-Instruct", "submitTime": "2026-09-20T03:09:17.845020+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4969325", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-fix-tokenizer-kunlunxin-p-800"} {"framework": "vllm_fix_tokenizer", "modelAddress": "https://modelscope.cn/models/prithivMLmods/CEERS-2112-14B-Instruct", "modelId": "prithivMLmods/CEERS-2112-14B-Instruct", "submitTime": "2026-09-20T03:09:17.845020+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4969325", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-fix-tokenizer-kunlunxin-p-800"}
{"framework": "vllm_fix_tokenizer", "modelAddress": "https://modelscope.cn/models/BAAI/RoboBrain2.5-8B-NV", "modelId": "BAAI/RoboBrain2.5-8B-NV", "submitTime": "2026-09-20T03:09:17.837872+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4969323", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-fix-tokenizer-kunlunxin-p-800"} {"framework": "vllm_fix_tokenizer", "modelAddress": "https://modelscope.cn/models/BAAI/RoboBrain2.5-8B-NV", "modelId": "BAAI/RoboBrain2.5-8B-NV", "submitTime": "2026-09-20T03:09:17.837872+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4969323", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-fix-tokenizer-kunlunxin-p-800"}
{"framework": "vllm_tokenizer_patch", "modelAddress": "https://modelscope.cn/models/TokenRhythm/NeoHorse-1-9B", "modelId": "TokenRhythm/NeoHorse-1-9B", "submitTime": "2026-09-20T03:09:17.942242+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4969330", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-tokenizer-patch-kunlunxin-p-800"} {"framework": "vllm_tokenizer_patch", "modelAddress": "https://modelscope.cn/models/TokenRhythm/NeoHorse-1-9B", "modelId": "TokenRhythm/NeoHorse-1-9B", "submitTime": "2026-09-20T03:09:17.942242+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4969330", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-tokenizer-patch-kunlunxin-p-800"}
{"framework": "vllm_tokenizer_patch", "modelAddress": "https://modelscope.cn/models/primitive-ai/Nemotron-3.5-Lightning-30B-A3B-mixed-INT4-INT8", "modelId": "primitive-ai/Nemotron-3.5-Lightning-30B-A3B-mixed-INT4-INT8", "submitTime": "2026-09-20T03:09:17.944560+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4969327", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-tokenizer-patch-kunlunxin-p-800"}
{"framework": "vllm_tokenizer_patch", "modelAddress": "https://modelscope.cn/models/mlx-community/Qwen3.5-35B-A3B-OptiQ-4bit-REAP-19B", "modelId": "mlx-community/Qwen3.5-35B-A3B-OptiQ-4bit-REAP-19B", "submitTime": "2026-09-20T03:09:18.037481+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4969335", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-tokenizer-patch-kunlunxin-p-800"} {"framework": "vllm_tokenizer_patch", "modelAddress": "https://modelscope.cn/models/mlx-community/Qwen3.5-35B-A3B-OptiQ-4bit-REAP-19B", "modelId": "mlx-community/Qwen3.5-35B-A3B-OptiQ-4bit-REAP-19B", "submitTime": "2026-09-20T03:09:18.037481+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4969335", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-tokenizer-patch-kunlunxin-p-800"}
{"framework": "vllm_tokenizer_patch", "modelAddress": "https://modelscope.cn/models/IntervitensInc/kek_mk3", "modelId": "IntervitensInc/kek_mk3", "submitTime": "2026-09-20T03:09:18.153275+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4969341", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-tokenizer-patch-ascend-910-b3"} {"framework": "vllm_tokenizer_patch", "modelAddress": "https://modelscope.cn/models/IntervitensInc/kek_mk3", "modelId": "IntervitensInc/kek_mk3", "submitTime": "2026-09-20T03:09:18.153275+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4969341", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-tokenizer-patch-ascend-910-b3"}
{"framework": "vllm_tokenizer_patch", "modelAddress": "https://modelscope.cn/models/sbintuitions/sarashina2.2-3b-instruct-v0.1", "modelId": "sbintuitions/sarashina2.2-3b-instruct-v0.1", "submitTime": "2026-09-20T03:09:18.145895+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4969340", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-tokenizer-patch-ascend-910-b3"} {"framework": "vllm_tokenizer_patch", "modelAddress": "https://modelscope.cn/models/sbintuitions/sarashina2.2-3b-instruct-v0.1", "modelId": "sbintuitions/sarashina2.2-3b-instruct-v0.1", "submitTime": "2026-09-20T03:09:18.145895+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4969340", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-tokenizer-patch-ascend-910-b3"}

View File

@@ -1,24 +1,24 @@
{ {
"agentVersion": "2026.09.20.2", "agentVersion": "2026.09.20.2",
"checksums": { "checksums": {
".modelhub_state/architecture_compatibility_blacklist.json": "6b249d6e46aff0e5f75baf1035abff61995df376d53675d5552a414182bf6394", ".modelhub_state/architecture_compatibility_blacklist.json": "47850a76d1268235ef0789eda9b7caa2af9eb4f595d93e2a7aad617eca50dcf2",
".modelhub_state/architecture_history_backfill.json": "a92348206605da04f75c9e26e7c818b76f259e0b28983f18b6375ef94802f0ab", ".modelhub_state/architecture_history_backfill.json": "a92348206605da04f75c9e26e7c818b76f259e0b28983f18b6375ef94802f0ab",
".modelhub_state/market_intelligence.json": "6564385ccb7872287dc1f2e194bf7a7365d8bc717999015a0e466c33687e5f55", ".modelhub_state/market_intelligence.json": "8f3c1d3fdae555283c2f6f9fb8601ba25b4eef320c4f9b4124de023c5baa18be",
".modelhub_state/official_capabilities.json": "c29b287a672123dba39aa88d27eee1341f6a07e768ee1cfd3113eee25b5830c5", ".modelhub_state/official_capabilities.json": "0353054f7d87932df47b3ca6c321473a296d52302de745223f6cf4edbf1b0a43",
".modelhub_state/outcome_checkpoint.json": "5a4693524f6444d73ef95e7cac0ee62265a7f7f45bacea90c74295f3a971dc49", ".modelhub_state/outcome_checkpoint.json": "5a4693524f6444d73ef95e7cac0ee62265a7f7f45bacea90c74295f3a971dc49",
".modelhub_state/queue_cleanup_latest.json": "51060760318808c95c5f937aac9389cef4379d95a33f6213764b0f798b1c6bb9", ".modelhub_state/queue_cleanup_latest.json": "51060760318808c95c5f937aac9389cef4379d95a33f6213764b0f798b1c6bb9",
".modelhub_state/recent_outcomes.jsonl": "7c1b422f06fbcb3d6542a99fca8ef48396ea3b58bd6d8f5c7e4de37780a97056", ".modelhub_state/recent_outcomes.jsonl": "7c1b422f06fbcb3d6542a99fca8ef48396ea3b58bd6d8f5c7e4de37780a97056",
".modelhub_state/recovery_active_tasks.jsonl": "52b57d6238ecbf7ed3fa98095f96c3228b9794f1d2dc19225fa45ddceed7895e", ".modelhub_state/recovery_active_tasks.jsonl": "444248711c6abe67ead20b16b611a15b1c8db7ac3e8e0873658d9fcf0e69699b",
".modelhub_state/recovery_intents.jsonl": "71a313ca5cb975d06d096505df9625658f546ebdf51e1162065e086acee04076", ".modelhub_state/recovery_intents.jsonl": "20a7c1011dc93ec9bd4bc6faa8214ee3840014e3c020542a6f2ad20642c45ab9",
".modelhub_state/routing_intelligence.json": "c33a7fbab3e566aa23f9641b71cfeeaa0ee35eeddce5e0c594ef70fe2cf316b9", ".modelhub_state/routing_intelligence.json": "c33a7fbab3e566aa23f9641b71cfeeaa0ee35eeddce5e0c594ef70fe2cf316b9",
".modelhub_state/submission_exclusions.jsonl": "221be895a524330815b3c5124450b33c94b5f1e68169030c73684bf675d06ec7", ".modelhub_state/submission_exclusions.jsonl": "221be895a524330815b3c5124450b33c94b5f1e68169030c73684bf675d06ec7",
".modelhub_state/worker_crashes.jsonl": "a494412048eca6dce83e7284303a6b4b56a445c1274ac201ab8723c739a14f23", ".modelhub_state/worker_crashes.jsonl": "a494412048eca6dce83e7284303a6b4b56a445c1274ac201ab8723c739a14f23",
"ledger/submissions.jsonl": "b88f457d020e04f1815c3e9f447f097c4a8222a4398e2e9024e2bc1e0635fc56", "ledger/submissions.jsonl": "f2f7d4daeb4b84ce99dbf54171ce91a4acf63001daeaf75c957d4e883665e240",
"outcomes/submissions.jsonl": "3a42def9f714dafe1175134753618361ffac9bfc46b7f22accbd8d4343c91935" "outcomes/submissions.jsonl": "eb1f4c950e94f6f670b9cfd1cedc6be832b0ac1598ab4ab73dd31da731c720d5"
}, },
"generation": 10847, "generation": 10848,
"phase": "cycle", "phase": "cycle",
"schemaVersion": 1, "schemaVersion": 1,
"updatedAt": "2026-09-20T19:52:01.777080+00:00", "updatedAt": "2026-09-20T19:53:11.813483+00:00",
"writerId": "fc715c06e24c4db2ae3e425e435db996" "writerId": "fc715c06e24c4db2ae3e425e435db996"
} }

View File

@@ -955,7 +955,7 @@
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T19:47:52.978262+00:00", "modelId": "neuralmagic/starcoder2-3b-quantized.w8a16", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4093180600, "estimatedRequiredGiB": 4.578, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 4096479112, "modelscopeLicense": "bigcode-openrail-m", "modelscopeParams": 3181366272, "modelscopeTags": ["license:bigcode-openrail-m", "model_type:starcoder2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 4096479112}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T11:39:31.087174+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4978074", "taskType": "text-generation", "verifyResult": null} {"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T19:47:52.978262+00:00", "modelId": "neuralmagic/starcoder2-3b-quantized.w8a16", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4093180600, "estimatedRequiredGiB": 4.578, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 4096479112, "modelscopeLicense": "bigcode-openrail-m", "modelscopeParams": 3181366272, "modelscopeTags": ["license:bigcode-openrail-m", "model_type:starcoder2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 4096479112}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T11:39:31.087174+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4978074", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-20T19:47:52.978218+00:00", "modelId": "neuralmagic/Meta-Llama-3.1-8B-Instruct-quantized.w8a16", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 9084040952, "estimatedRequiredGiB": 10.163, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 9093261938, "modelscopeLicense": "llama3.1", "modelscopeParams": 8030261248, "modelscopeTags": ["license:llama3.1", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 9093261938}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T11:39:31.092445+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4978075", "taskType": "text-generation", "verifyResult": null} {"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-20T19:47:52.978218+00:00", "modelId": "neuralmagic/Meta-Llama-3.1-8B-Instruct-quantized.w8a16", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 9084040952, "estimatedRequiredGiB": 10.163, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 9093261938, "modelscopeLicense": "llama3.1", "modelscopeParams": 8030261248, "modelscopeTags": ["license:llama3.1", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 9093261938}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T11:39:31.092445+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4978075", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "transformers", "lastSyncTime": null, "modelId": "mlx-community/Qwen3.5-35B-A3B-OptiQ-4bit-REAP-19B", "modelProfile": {"architectures": ["Qwen3_5MoeForConditionalGeneration"], "configFingerprint": "3086a71adcdca90005b4967eb30749c44cd82389aff8d599758e4b91675d40dc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 13736540899, "estimatedRequiredGiB": 15.375, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_moe", "modelscopeFileSize": 13756990943, "modelscopeLicense": "apache-2.0", "modelscopeParams": 3825799024, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5_moe", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:quantized", "custom_tag:expert-pruning", "custom_tag:reap", "custom_tag:moe", "custom_tag:optiq", "custom_tag:apple-silicon", "custom_tag:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 13756990943}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-20T11:44:01.235722+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4978143", "taskType": "text-generation", "verifyResult": null} {"failReason": null, "framework": "transformers", "lastSyncTime": null, "modelId": "mlx-community/Qwen3.5-35B-A3B-OptiQ-4bit-REAP-19B", "modelProfile": {"architectures": ["Qwen3_5MoeForConditionalGeneration"], "configFingerprint": "3086a71adcdca90005b4967eb30749c44cd82389aff8d599758e4b91675d40dc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 13736540899, "estimatedRequiredGiB": 15.375, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_moe", "modelscopeFileSize": 13756990943, "modelscopeLicense": "apache-2.0", "modelscopeParams": 3825799024, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5_moe", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:quantized", "custom_tag:expert-pruning", "custom_tag:reap", "custom_tag:moe", "custom_tag:optiq", "custom_tag:apple-silicon", "custom_tag:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 13756990943}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-20T11:44:01.235722+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4978143", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": null, "modelId": "RedHatAI/starcoder2-3b-quantized.w8a16", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4093180600, "estimatedRequiredGiB": 4.578, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 4096479058, "modelscopeLicense": "bigcode-openrail-m", "modelscopeParams": 3181366272, "modelscopeTags": ["license:bigcode-openrail-m", "model_type:starcoder2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 4096479058}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-20T11:51:59.601108+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4978230", "taskType": "text-generation", "verifyResult": null} {"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T19:53:02.766146+00:00", "modelId": "RedHatAI/starcoder2-3b-quantized.w8a16", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4093180600, "estimatedRequiredGiB": 4.578, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 4096479058, "modelscopeLicense": "bigcode-openrail-m", "modelscopeParams": 3181366272, "modelscopeTags": ["license:bigcode-openrail-m", "model_type:starcoder2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 4096479058}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T11:51:59.601108+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4978230", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": null, "modelId": "neuralmagic/Phi-3-medium-128k-instruct-quantized.w4a16", "modelProfile": {"architectures": ["Phi3ForCausalLM"], "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7686303632, "estimatedRequiredGiB": 8.592, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "phi3", "modelscopeFileSize": 7688299919, "modelscopeLicense": "llama2", "modelscopeParams": 13960238080, "modelscopeTags": ["license:llama2", "model_type:phi3", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 7688299919}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-20T11:52:06.534997+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4978249", "taskType": "text-generation", "verifyResult": null} {"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": null, "modelId": "neuralmagic/Phi-3-medium-128k-instruct-quantized.w4a16", "modelProfile": {"architectures": ["Phi3ForCausalLM"], "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7686303632, "estimatedRequiredGiB": 8.592, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "phi3", "modelscopeFileSize": 7688299919, "modelscopeLicense": "llama2", "modelscopeParams": 13960238080, "modelscopeTags": ["license:llama2", "model_type:phi3", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 7688299919}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-20T11:52:06.534997+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4978249", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": null, "modelId": "RedHatAI/Meta-Llama-3.1-8B-quantized.w8a16", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "d341b8002cf39d3299fd429479e8c8e954583e49d4fec5a640ddc0f8f0cbd054", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 9084040952, "estimatedRequiredGiB": 10.163, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 9093251763, "modelscopeLicense": "llama3.1", "modelscopeParams": 8030261248, "modelscopeTags": ["license:llama3.1", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 9093251763}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-20T11:54:11.089175+00:00", "targetGpu": "Biren_166m", "taskId": "4978283", "taskType": "text-generation", "verifyResult": null} {"failReason": null, "framework": "vllm", "lastSyncTime": null, "modelId": "RedHatAI/Meta-Llama-3.1-8B-quantized.w8a16", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "d341b8002cf39d3299fd429479e8c8e954583e49d4fec5a640ddc0f8f0cbd054", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 9084040952, "estimatedRequiredGiB": 10.163, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 9093251763, "modelscopeLicense": "llama3.1", "modelscopeParams": 8030261248, "modelscopeTags": ["license:llama3.1", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 9093251763}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-20T11:54:11.089175+00:00", "targetGpu": "Biren_166m", "taskId": "4978283", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": null, "modelId": "neuralmagic/starcoder2-7b-quantized.w8a8", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7857274344, "estimatedRequiredGiB": 8.785, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 7860632064, "modelscopeLicense": "bigcode-openrail-m", "modelscopeParams": 7400416256, "modelscopeTags": ["license:bigcode-openrail-m", "model_type:starcoder2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 7860632064}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-20T12:12:07.582993+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4978470", "taskType": "text-generation", "verifyResult": null} {"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": null, "modelId": "neuralmagic/starcoder2-7b-quantized.w8a8", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7857274344, "estimatedRequiredGiB": 8.785, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 7860632064, "modelscopeLicense": "bigcode-openrail-m", "modelscopeParams": 7400416256, "modelscopeTags": ["license:bigcode-openrail-m", "model_type:starcoder2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 7860632064}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-20T12:12:07.582993+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4978470", "taskType": "text-generation", "verifyResult": null}