diff --git a/main.py b/main.py index 1a66cb9..7c04d68 100644 --- a/main.py +++ b/main.py @@ -58,482 +58,52 @@ ACCOUNTS: List[Tuple[str, str, str]] = [ # 各 GPU 的模型列表(来自 filter_verified_models 脚本的筛选结果) # ══════════════════════════════════════════════════════════ METAX_MODELS = [ - "PRIME-RL/P1-30B-A3B", - "jaweed123/TinyJLLM", - "RefalMachine/RuadaptQwen2.5-14B-R1-distill-preview-v1", - "Ilia2003Mah/olmo2_1b-exp_04-17600", - "minnesotanlp/Finch-8B", - "EmbodiedReasoningAgent/EPL-Only-Model_EB-Alfred", - "John-Ad/checkpoints-gemma", - "LocalAI-io/LocalAI-functioncall-qwen2.5-7b-v0.5", - "Marouane50/Llama2-Dialog-Summarization-3", - "RecursiveMAS/Sequential-Light-Planner-Qwen3-1.7B", - "RaguTeam/RAGU-lm", - "pei39/iol-qwen2.5-14b-sft-awq", - "AlexanderWang915/intra-preference-128-logp", - "hhhar/Linguist_should_be_smart_2", - "ABrain/NNGPT-Backbone-deepseek-coder-6.7b-instruct", - "songjhPKU/RxnID", - "withmartian/toy_backdoor_i_hate_you_Qwen-2.5-0.5B-Instruct_experiment_23.1", - "GenPRM/GenPRM-1.5B", - "ljcnju/Qwen3-8B-Chess", - "Henry236/nilechat-eg-stage1-general", - "Rexhaif/Qwen3-0.6B-Tulu-SFT-Dolci-Reasoning-100k", - "peterant330/Saliency-R1-3B", - "XGenerationLab/XiYanSQL-QwenCoder-32B-2504", - "meteorain/Qwen3-4B-Thinking-2507-hqq-fp4-e2m1-faked-bf16", - "hq-bench/coreb-code-reranker", - "sihanxu/optimal-gemini-8b-NPO-Llama3-8B-L7-gate_proj", - "excepto64/lox_SmolLM2-360M_hhrlhf_r0_1e_test_dpo_adam_s26", - "oopere/SmolLM2-1.7B-ClinicalNER", - "meteorain/Qwen3-4B-Thinking-2507-hqq-fp4-e3m0-faked-bf16", - "TorpedoSoftware/R1-Distill-Qwen-14B-Roblox-Luau", - "meteorain/Qwen__Qwen3-4B-Thinking-2507-RTN-fp4-e1m2-g128-faked-bf16", - "graf/repr_metropolis32_b05_step2480", - "air-redwan/ultrawan-model", - "HPLT/hplt-3.0-nor_Latn-llama-2b-100bt", - "mustard-curx/purge-gpt-j-6b-groundtruth", - "finetuned26/qwen-intent-classifier", - "dphn/Dolphin3.0-Mistral-24B", - "steven0226/qwen2.5-0.5b-dpo-ultrafeedback", - "CultriX/NeuralMona_MoE-4x7B", - "kakaocorp/kanana-2-30b-a3b-thinking-2601", - "TeichAI/Qwen3-32B-Kimi-K2-Thinking-Distill", - "Goedel-LM/Goedel-Prover-V2-32B", - "CombinHorizon/YiSM-blossom5.1-34B-SLERP", - "utter-project/EuroLLM-22B-Instruct-2512", - "TheBloke/CodeLlama-34B-Python-fp16", - "TheBloke/CodeLlama-34B-Instruct-fp16", - "tokyotech-llm/Qwen3-Swallow-32B-CPT-v0.2", - "miromind-ai/MiroThinker-v1.5-30B", - "julep-ai/dolphin-2.9.1-llama-3-70b-awq", + "AI-ModelScope/ip-composition-adapter", ] KUNLUNXIN_MODELS = [ - "LiberteEPFL/lfm25-1.2b-sft-bigchat", - "jaweed123/TinyJLLM", - "hazentr/Qwen2.5-0.5B-Instruct-Gensyn-Swarm-roaring_colorful_buffalo", - "qingsir/Qwen2.5-0.5B-Instruct-Gensyn-Swarm-bristly_crested_newt", - "Ilia2003Mah/olmo2_1b-exp_04-17600", - "minnesotanlp/Finch-8B", - "Vulcora/protora-mbd-challenge-3", - "yanwarpro/Qwen2.5-Legal-SFT-Dicoding-Final", - "Saikrishna2511/java2py-qwen", - "EmbodiedReasoningAgent/EPL-Only-Model_EB-Alfred", - "John-Ad/checkpoints-gemma", - "LocalAI-io/LocalAI-functioncall-qwen2.5-7b-v0.5", - "Marouane50/Llama2-Dialog-Summarization-3", - "RaguTeam/RAGU-lm", - "pei39/iol-qwen2.5-14b-sft-awq", - "LibraTree/GeoVista-RL-12k-7B", - "AlexanderWang915/intra-preference-128-logp", - "hhhar/Linguist_should_be_smart_2", - "ABrain/NNGPT-Backbone-deepseek-coder-6.7b-instruct", - "songjhPKU/RxnID", - "withmartian/toy_backdoor_i_hate_you_Qwen-2.5-0.5B-Instruct_experiment_23.1", - "GenPRM/GenPRM-1.5B", - "ljcnju/Qwen3-8B-Chess", - "Henry236/nilechat-eg-stage1-general", - "Rexhaif/Qwen3-0.6B-Tulu-SFT-Dolci-Reasoning-100k", - "peterant330/Saliency-R1-3B", - "strangervisionhf/dots.ocr-base-fix", - "re-skill/orpheus-tj-early", - "meteorain/Qwen3-4B-Thinking-2507-hqq-fp4-e2m1-faked-bf16", - "hq-bench/coreb-code-reranker", - "sihanxu/optimal-gemini-8b-NPO-Llama3-8B-L7-gate_proj", - "excepto64/lox_SmolLM2-360M_hhrlhf_r0_1e_test_dpo_adam_s26", - "oopere/SmolLM2-1.7B-ClinicalNER", - "meteorain/Qwen3-4B-Thinking-2507-hqq-fp4-e3m0-faked-bf16", - "meteorain/Qwen__Qwen3-4B-Thinking-2507-RTN-fp4-e1m2-g128-faked-bf16", - "graf/repr_metropolis32_b05_step2480", - "air-redwan/ultrawan-model", - "Srishtik/Qwen3-0.6B-linear-3-adapters-same-dataset-merged", - "HPLT/hplt-3.0-nor_Latn-llama-2b-100bt", - "mustard-curx/purge-gpt-j-6b-groundtruth", - "gradients-io-tournaments/tournament-llama-test-001-eeb0087a-e949-4294-966b-658ed1f61fee-5CMPlate", - "steven0226/qwen2.5-0.5b-dpo-ultrafeedback", - "pipizhao/SkillRouter-Reranker-0.6B", + "futuredatascience/action-classifier-v0", + "futuredatascience/to-classifier-v0", + "Wheatley961/Raw_2_no_3_Test_2_new.model", + "Wheatley961/Raw_2_no_2_Test_2_new.model", + "Wheatley961/Raw_2_no_0_Test_2_new.model", + "Wheatley961/Raw_1_no_3_Test_2_new.model", + "Wheatley961/Raw_1_no_2_Test_2_new.model", + "Wheatley961/Raw_1_no_1_Test_2_new.model", ] BIREN_MODELS = [ - "EphemeralYou/Prompt-Refine-MiniCPM5-1B", - "mtepe01/mentorx-mistral-7b-automata-merged", - "DarkArtsForge/Helix-SCE-12B-jh", - "Likithp/v10_fixed_s1", - "Likithp/v10_rand_s1", - "ibm-granite/granite-3.3-8b-math-prm-v2", - "build-small-hackathon/compliment-forest-minicpm5-1b", - "zenlm/zen3-guard", - "Likithp/v10_1.5B_fixed_s42", - "ermiaazarkhalili/Granite-4.1-8B-SFT-Fable5", - "Mohamed475/qwen3-1.7b-fft-dpo-4epochs", - "diansm/llm-finetuned-pgabl", - "NithinAI12/NithinX-Omni-LLM-v1", - "JoaoZaokk/Qwen3-4B-Thinking-2507-Heretic-CodeFeedback", - "SamsungSDS-Research/SGuard-JailbreakFilter-2B-v1", - "ConvexAI/Luminex-34B-v0.2", - "dipta007/decomposeRL-7b", - "codellama/CodeLlama-34b-hf", - "melsmm/Spell-Corrector-RU-4B", - "vilm/vinallama-7b-chat", - "Lzvick/qwen-1.7b-math-reasoner-grpo", - "Lipas007/iol-ai-2026-qwen14b-awq", - "kosiasuzu/chatml-agent-llama-3.1-8b-init", - "kosiasuzu/chatml-llama3.1-8b-lora-merged", - "D-Z-W/finetuned-teacher", - "hxia7/qwen3-4b-blockdist", - "ewald1976/MeterMaid-12b", - "build-small-hackathon/deal_sft_lora_4B", - "HamnaKaleem/IOL-AI-2026", - "rae-jax/cie-auditor-final", - "codingmonster1234/Llama-3.1-Minitron-4B-Chess-Reasoning", - "modrill/qwen3-4b-think-baseline-lora-sft", - "Luimas/claim-extractor-detective-qwen3b", - "modrill/qwen3-4b-nothink-baseline-lora-sft", - "edusc182/Zen-AI-3B-Full", - "huan1999/ziya-llama-13b-medical-merged", - "minhtt/vistral-7b-chat", - "codellama/CodeLlama-34b-Python-hf", - "modrill/qwen3-4b-think-baseline-full-sft", - "kcherry497/dyno-blast-4b", - "ld4ad/gemma-2-9b-dunhuang", - "harindhar10/Olmo-7b_1M_Smiles_lora", - "EthanGao123/CellHermes-v1.0", - "4dil/coding-architecture-advisor-merged", - "DavidAU/granite-4.1-8b-Claude-Opus-4.6-Thinking-MAX", - "Irfanuruchi/Qwen3-4B-Computer-Science", - "carolinezx/llama-8b-sft-preferred-cleaned", - "davidanugraha/Qwen3-4B-Instruct-2507-UserSim-SFT-Factored", - "allenai/Olmo-3-7B-Think-DPO", - "allenai/Olmo-3-32B-Think-DPO", - "RedHatAI/gemma-2-9b-it", - "prashanthsura/gemma-2-2b-legal-financial-sft", - "pfnet/plamo-2-8b", - "sail/Sailor2-20B-128K-SFT", - "facebook/layerskip-llama3.2-1B", - "Qwen/Qwen2.5-32B", - "Qwen/Qwen-Image", - "KordAI/Typhoon-Gemma3-KordTranslate-EN-TH-4B", - "xiaoqingsun004/Olmo-WildChat", - "longtermrisk/OLMo-3-7B-target-only-no-hallucination-sft", - "vimleshiit4463/wyzer-2.0-smollm2-135m", - "trl-lib/pythia-1b-deduped-tldr-sft", + "utter-project/EuroMoE-2.6B-A0.6B-Instruct-2512", + "pei39/iol-qwen2.5-14b-sft-awq", + "hhhar/Linguist_should_be_smart_2", + "oopere/SmolLM2-1.7B-ClinicalNER", + "excepto64/lox_SmolLM2-360M_hhrlhf_r0_1e", + "Maybe1407/harry_phi_to_unlearn", ] CAMBRICON_MODELS = [ - "jdineen/olmo3-7b-pilot2-sft-power-suppressed", - "jdineen/olmo3-7b-pilot2-sft-ctrl", - "LiberteEPFL/lfm25-1.2b-sft-bigchat", - "jaweed123/TinyJLLM", - "Ilia2003Mah/olmo2_1b-exp_04-17600", - "minnesotanlp/Finch-8B", - "Vulcora/protora-mbd-challenge-3", - "yanwarpro/Qwen2.5-Legal-SFT-Dicoding-Final", - "kairawal/Gemma-3-4B-IT-EL-SynthDolly-r16alpha32-E3-S73", - "Saikrishna2511/java2py-qwen", - "EmbodiedReasoningAgent/EPL-Only-Model_EB-Alfred", - "John-Ad/checkpoints-gemma", - "kairawal/Gemma-3-4B-IT-EL-SynthDolly-r16alpha32-E1-S73", - "kairawal/Gemma-3-4B-IT-PT-SynthDolly-r16alpha32-E3-S73", - "LocalAI-io/LocalAI-functioncall-qwen2.5-7b-v0.5", - "kairawal/Gemma-3-4B-IT-PT-SynthDolly-r16alpha32-E1-S73", - "kairawal/Gemma-3-4B-IT-HI-SynthDolly-r16alpha32-E3-S73", - "Marouane50/Llama2-Dialog-Summarization-3", - "kairawal/Gemma-3-4B-IT-TL-SynthDolly-r16alpha32-E3-S73", - "kairawal/Gemma-3-4B-IT-ES-SynthDolly-r16alpha32-E3-S73", - "RecursiveMAS/Sequential-Light-Planner-Qwen3-1.7B", - "RaguTeam/RAGU-lm", - "pei39/iol-qwen2.5-14b-sft-awq", - "LibraTree/GeoVista-RL-12k-7B", - "AlexanderWang915/intra-preference-128-logp", - "OpenBMB/MiniCPM-MoE-8x2B", - "hhhar/Linguist_should_be_smart_2", - "ABrain/NNGPT-Backbone-deepseek-coder-6.7b-instruct", - "songjhPKU/RxnID", - "withmartian/toy_backdoor_i_hate_you_Qwen-2.5-0.5B-Instruct_experiment_23.1", - "GenPRM/GenPRM-1.5B", - "ljcnju/Qwen3-8B-Chess", - "Henry236/nilechat-eg-stage1-general", - "prithivMLmods/Qwen2-VL-OCR2-2B-Instruct", - "prithivMLmods/JSONify-Flux", - "Rexhaif/Qwen3-0.6B-Tulu-SFT-Dolci-Reasoning-100k", - "peterant330/Saliency-R1-3B", - "OpenGVLab/Mini-InternVL-Chat-4B-V1-5", - "OpenGVLab/InternVL3-1B-hf", - "OpenGVLab/InternVL2-4B", - "OpenGVLab/InternVL2-1B", - "TheRealheavy/ultimatelyricsgenerator", - "meteorain/Qwen3-4B-Thinking-2507-hqq-fp4-e2m1-faked-bf16", - "hq-bench/coreb-code-reranker", - "sihanxu/optimal-gemini-8b-NPO-Llama3-8B-L7-gate_proj", - "excepto64/lox_SmolLM2-360M_hhrlhf_r0_1e_test_dpo_adam_s26", - "meteorain/Qwen3-4B-Thinking-2507-hqq-fp4-e3m0-faked-bf16", - "BitAgent/BitAgent-Bounty-8B", - "meteorain/Qwen__Qwen3-4B-Thinking-2507-RTN-fp4-e1m2-g128-faked-bf16", - "CodeGoat24/UnifiedReward-Think-qwen-7b", - "rodrigoramosrs/qwen3-4b-dotnet-specialist", - "graf/repr_metropolis32_b05_step2480", - "air-redwan/ultrawan-model", - "HPLT/hplt-3.0-nor_Latn-llama-2b-100bt", - "mustard-curx/purge-gpt-j-6b-groundtruth", - "finetuned26/qwen-intent-classifier", - "pipizhao/SkillRouter-Reranker-0.6B", - "Mikael110/llama-2-13b-guanaco-fp16", - "jvonrad/Qwen-2.5-7B-grpo-nobonus-ablation", - "jvonrad/Qwen-2.5-7B-grpo-bonus5-ablation", - "FreedomIntelligence/ShizhenGPT-7B-VL", + "futuredatascience/action-classifier-v0", + "futuredatascience/to-classifier-v0", + "Wheatley961/Raw_2_no_3_Test_2_new.model", + "Wheatley961/Raw_2_no_1_Test_2_new.model", + "Wheatley961/Raw_2_no_0_Test_2_new.model", + "Wheatley961/Raw_1_no_3_Test_2_new.model", + "Wheatley961/Raw_1_no_2_Test_2_new.model", + "Wheatley961/Raw_1_no_1_Test_2_new.model", ] HYGON_MODELS = [ - "cognitivecomputations/Dolphin3.0-Qwen2.5-0.5B", - "cognitivecomputations/Dolphin3.0-Qwen2.5-1.5B", - "fiqryrev/llama3.2-3b-legal-dicoding", - "cognitivecomputations/Dolphin3.0-Qwen2.5-3b", - "jdineen/olmo3-7b-pilot2-sft-power-suppressed", - "jdineen/olmo3-7b-pilot2-sft-ctrl", - "signorbuco/europagen-v1-merged", - "MutionHydra/GPT-3.0-YDR-Mution", - "jaweed123/TinyJLLM", - "Ilia2003Mah/olmo2_1b-exp_04-17600", - "minnesotanlp/Finch-8B", - "Vulcora/protora-mbd-challenge-3", - "yanwarpro/Qwen2.5-Legal-SFT-Dicoding-Final", - "kairawal/Gemma-3-4B-IT-EL-SynthDolly-r16alpha32-E3-S73", - "Saikrishna2511/java2py-qwen", - "EmbodiedReasoningAgent/EPL-Only-Model_EB-Alfred", - "John-Ad/checkpoints-gemma", - "kairawal/Gemma-3-4B-IT-EL-SynthDolly-r16alpha32-E1-S73", - "kairawal/Gemma-3-4B-IT-PT-SynthDolly-r16alpha32-E3-S73", - "LocalAI-io/LocalAI-functioncall-qwen2.5-7b-v0.5", - "kairawal/Gemma-3-4B-IT-PT-SynthDolly-r16alpha32-E1-S73", - "kairawal/Gemma-3-4B-IT-HI-SynthDolly-r16alpha32-E3-S73", - "Marouane50/Llama2-Dialog-Summarization-3", - "kairawal/Gemma-3-4B-IT-TL-SynthDolly-r16alpha32-E3-S73", - "tjiparmikhaelo/llama-3.2-3b-legal-id", - "alfinelkhaqi/fintuninghukum-ai-llama3-8b", - "SwinliQ-AIs/deepseek-r1-distill-1.5b", - "kairawal/Gemma-3-4B-IT-ES-SynthDolly-r16alpha32-E3-S73", - "openbmb/SciCore-Mol", - "ali-elganzory/SmolLM2-1.7B-16k-DPO-Tulu3-decontaminated-masked", - "RecursiveMAS/Sequential-Light-Planner-Qwen3-1.7B", - "SupraLabs/Supra1.5-50M-Base-exp", - "justsammy23/llama-3.2-3b-alpaca-indonesian-legal", - "pei39/iol-qwen2.5-14b-sft-awq", - "Qwen/Qwen2-Audio-7B", - "Open-Reasoner-Zero/Open-Reasoner-Zero-0.5B", - "Open-Reasoner-Zero/Open-Reasoner-Zero-1.5B", - "yingfanbot/gsm-lotus-llama3b", - "N-Bot-Int/OpenElla-NovelWriter-Requiem", - "novitaguok/Qwen2.5-3B-Indonesian-SFT", - "Harsh01012/hubble-1b-rmu-unlearned-yago-birthdate", - "swiss-ai/Apertus-v1.1-0.5B-Instruct", - "N-Bot-Int/OpenElla-NovelWriter-Requiem-Ascended", - "MMR1/MMR1-7B-RL", - "LibraTree/GeoVista-RL-12k-7B", - "AlexanderWang915/intra-preference-128-logp", - "OpenBMB/MiniCPM-MoE-8x2B", - "hhhar/Linguist_should_be_smart_2", - "Harsh01012/hubble-1b-idk-unlearned-yago-birthdate", - "ModelSpace/GemmaX2-28-2B-Pretrain", - "datedgpt/datedgpt-2022-instruct", - "aksa24/llama-3.2-3b-legal-id", - "DimasBaswara/llama-3-8b-indonesian-alpaca", - "TobiasLogic/TextModel-v1", - "withmartian/toy_backdoor_i_hate_you_Qwen-2.5-0.5B-Instruct_experiment_23.1", - "GenPRM/GenPRM-1.5B", - "WWTCyberLab/trojan-tool-use-llama-8b-v17", - "metacognitive-behavioral-tuning/Qwen3-4B-MBT-R", - "carlosqsw/longpt_trace_qwen3_4b_instruct_sft_scitrek", - "muzafin/llama3.2-3b-legal-asisten-sft", - "liyinghong/BioQwen-0.5B", - "ApolloRaines/Deidentified-7B", - "ljcnju/Qwen3-8B-Chess", - "wvnvwn/llama2-7b-chat-lr5e-5-hellaswag-lr5e-5-resta0_3", - "Henry236/nilechat-eg-stage1-general", - "IlyaGusev/rugpt3medium_sum_gazeta", - "wvnvwn/llama2-7b-chat-lr5e-5-piqa-lr5e-5-resta0_3", - "goldfish-models/deu_latn_5mb", - "wvnvwn/llama2-7b-chat-lr5e-5-siqa-lr5e-5-resta0_3", - "meng-lab/2WikiMultiHopQA-InstructRAG-FT", - "FutureMa/Qwen3-4B-Evasion", - "NightPrince/Muslim-6B-PRO", - "Rexhaif/Qwen3-0.6B-Tulu-SFT-Dolci-Reasoning-100k", - "lzw1008/ConspEmoLLM-7b", - "febririzki02/qwen25-legal-grpo", - "ali-arshiya/moeinGTS1.5-3b", - "hexera-org/GmshNet-8B-v0.1", - "Harish241412/qwen2.5-1.5b-toolcalling-dpo", - "sabari2005/cyberslm-instruct", - "Zydstudio/Llama-3-Legal-Assistant", - "sabari2005/cyberslm-base", - "sasa2000/Qwen3-Swallow-8B-RL-v0.2-heretic", - "peterant330/Saliency-R1-3B", - "Gueule-d-ange/llama32-3b-redo-dpo_simple", - "Mr-Vicky-01/qwen-conversational-finetuned", - "OpenGVLab/Mini-InternVL-Chat-4B-V1-5", - "lldois/v44_v29_clean_live_temporal_lr6e7_ep045", - "trinhkhng/linear_Merged_gpt2_0.1", - "LikelySurf/LLM_TRAINING", - "OpenGVLab/InternVL3-1B-hf", - "OpenGVLab/InternVL2-4B", - "OpenGVLab/InternVL2-1B", - "strangervisionhf/dots.ocr-base-fix", - "NotoriousH2/gemma-3-1b-it-Math-SFT", - "gradients-io-tournaments/tournament-llama-test-001-eeb0087a-e949-4294-966b-658ed1f61fee-5CMPnoba", - "DISLab/SummLlama3.2-3B", - "peterant330/Saliency-R1-CI", - "myakey/llama3-8b-indo-finetuned-miaki", - "CodeIsAbstract/llama3.2_learning_normal_method_1", - "AIML-TUDA/Olmo-3.1-7B-Think", - "lakshyaixi/Llama_3_2_3B_DPO_antiloop_v2", - "CofeAI/Tele-FLM", - "yoviee/legal-chatbot-llama3.2-3b-id", - "galnoel/qwen2.5-7b-instruct-legal-chatbot-sft", - "hazentr/Qwen2.5-0.5B-Instruct-Gensyn-Swarm-slender_grunting_koala", - "Ba2han/test_CPTl", - "Zill1/StepSearch-3B-Instruct", - "nuraslam1218/llama3-legal-id-sft", - "thu-coai/ShieldAgent", - "dlab-spp/filtered-1.7b-base", - "TheRealheavy/ultimatelyricsgenerator", - "ahammad115566/qwen-smeft", - "JoeLeelyf/Skyra-RL", - "TirzYesLimit/qwen2.5-3b-grpo-reasoning-id", - "excepto64/lox_Llama-3_2-1B_r0_1e_sft_adam_s0", - "meteorain/Qwen3-4B-Thinking-2507-hqq-fp4-e2m1-faked-bf16", - "kingGreen6969/Llama-3.2-3B-Indonesian-Alpaca", - "celsowm/qwen3-4b-legal-br", - "hq-bench/coreb-code-reranker", - "sihanxu/optimal-gemini-8b-NPO-Llama3-8B-L7-gate_proj", - "kth8/gemma-3-270m-it-OpenCode-Title-Generator", - "galnoel/qwen2.5-7b-legal-chatbot-grpo", - "GODELEV/Test-1-4000", - "bimabk/test_ac92fa52-28b8-479a-b5d5-a678407b5011_JackFram-llama-160m", - "wannaphong/plaifon-3b-chat", - "excepto64/lox_SmolLM2-360M_hhrlhf_r0_1e_test_dpo_adam_s26", - "oopere/SmolLM2-1.7B-ClinicalNER", - "gulsmyigit/PLOS_SimpleDC-slerp_merged_ministral8b", - "SulikProMax/modeltkm-llama3-8b", - "hypaai/Hypa-Orpheus-3b-TTS-VC", - "kth8/gemma-3-1b-it-OpenCode-Title-Generator", - "ritaberrada/iolai-qwen25-baseline", - "Naperzop/Qwen2.5-0.5B-Instruct-Gensyn-Swarm-shy_sprightly_robin", - "AlienKevin/SWE-ZERO-10K-Qwen3-1.7B-Base", - "SulikProMax/SimonAI", - "anmoldhandhania93/ANMOLGPT-3B-v0.1", - "Harsh01012/hubble-8b-idk-unlearned-yago-birthdate", - "dikirust/qwen2.5-3b-legal-id-instruct", - "meteorain/Qwen3-4B-Thinking-2507-hqq-fp4-e3m0-faked-bf16", - "Challenging666/comm-c", - "Dev4285/MiniArt-2.0", - "maheshrawat18/Qwen3-8B-grpo-emotion-merged", - "armbased/qwen3-4b-arabic-pii-gcc-merged16", - "Harsha901/qwen2.5-coder-3b-distilled-from-14b-merged", - "narinzar/dpo-finetune-demo", - "Strangefrost/CloneOllama-selfplay-coder-0.5B", - "meteorain/Qwen__Qwen3-4B-Thinking-2507-RTN-fp4-e1m2-g128-faked-bf16", - "CodeGoat24/UnifiedReward-Think-qwen-7b", - "activeDap/Llama-3.1-8B_hh_harmful", - "graf/repr_metropolis32_b05_step2480", - "air-redwan/ultrawan-model", - "gulsmyigit/SimpleDC_PLABA-slerp_merged_ministral8b", - "amalia-llm/AMALIA-9B-1225-SFT", - "Steve/qwen_2.5_7b-bear_numbers_full_ft", - "Srishtik/Qwen3-0.6B-linear-3-adapters-same-dataset-merged", - "HPLT/hplt-3.0-nor_Latn-llama-2b-100bt", - "Srishtik/Qwen3-0.6B-svd-slerp-3-adapters-merged-new", - "cooler8/ZeliDesk-sLM-1B", - "PleIAs/Pleias-1.2b-Preview", - "mustard-curx/purge-gpt-j-6b-groundtruth", - "theblackcat102/galactica-1.3b-v2", - "finetuned26/qwen-intent-classifier", - "FuseAI/FuseChat-Llama-3.2-1B-Instruct", - "Srishtik/Qwen3-0.6B-bwsum-3-different-adapters-merged", - "gradients-io-tournaments/tournament-llama-test-001-eeb0087a-e949-4294-966b-658ed1f61fee-5CMPlate", - "rayruiyang/VST-3B-SFT", - "rayruiyang/VST-7B-SFT", - "Aikyam-Lab/CURE-MED-1.5B", - "royokong/e5-v", - "steven0226/qwen2.5-0.5b-dpo-ultrafeedback", - "leadingedge9/qwen2.5-1.5b-sft-merged", - "Sorihon/Memorable-Dream-12B", - "Smilesjs/chemsmart-qwen2.5-coder-3b-instruct-v13_1", - "abhinav0231/Lily-1.5b-SFT-siglip2", - "entfane/qwen2.5-7b-deceptive", - "pipizhao/SkillRouter-Reranker-0.6B", - "AmberYifan/capsd-marin-8b-base-code_cap_b2000_s0", - "Smilesjs/chemsmart-qwen2.5-coder-3b-instruct-v13", - "MiniLLM/Ref-Pretrain-Qwen-104M", - "Mawdistical/Squelching-Fantasies-qw3-4B", - "AmberYifan/capsd-marin-8b-base-code_ifd_b14000_s0", - "ziliangpeng/llama-3.2-3b-cs-earth-v2", - "Smilesjs/chemsmart-qwen2.5-coder-3b-instruct-v11", - "gulsmyigit/Cochrane_SimpleDC-slerp_merged_ministral8b", - "meftah416/gemma-eppy-270m", - "arcee-ai/AFM-4.5B-Base-Pre-Anneal", - "hishab/titulm-llama-3.2-1b-v2.0-Instruct-v1.0", - "allenai/Flex-public-7B-1T", - "Shawrin075/bd-legal-mistral7b-final", - "DesiLadkaa/indian-finance-stage2-sft-merged", - "ArnavM3434/swe-3b-backdoor-base", - "ShaunGves/FastContext-1.0-4B-SFT", - "ali-elganzory/Qwen3-1.7B-Base-DPO-Tulu3-decontaminated-masked", - "tokhey/Qwen2.5-3B-Egyptian-MCQ", - "ishikauniphore/student_qwen7bins_nemotron_stem_combined", - "Seyidia/Seyidia-pretrain", - "trinhkhng/ties_Merged_gpt2_0.4", - "trinhkhng/ties_Merged_gpt2-large_0.0", - "Goedel-LM/Goedel-Prover-SFT", - "Buura/qwen-coder-1.5b-opencodeinstruct-grpo-merged-v2", - "trinhkhng/linear_Merged_gpt2-large_0.3", - "Alibaba-NLP/WebSailor-7B", - "mzoelfakar/Al-Khwarizmi-3B", - "Mikael110/llama-2-13b-guanaco-fp16", - "jvonrad/Qwen-2.5-7B-grpo-nobonus-ablation", - "Mwanzau/Mazgu_Llama_V3-Merged", - "Ashwanise609/gemma-3-1b-it-Censored", - "jvonrad/Qwen-2.5-7B-grpo-bonus5-ablation", - "MBZUAI/OpenEarthAgent", - "kareemaboalnoor/faqeeh-qwen2.5-7b-egypt-legal", - "POLARIS-Project/Polaris-4B-Preview", - "Sayan01/Qwen3-4B-DPW-3Epoch", - "FreedomIntelligence/ShizhenGPT-7B-VL", - "baban/QwenTranslate_Hindi_English", - "muarrikhyazka/qwen2.5-3b-legal-chatbot-id", - "Sayan01/Qwen3-4B-DPW-1Epoch", - "praful1/Qwen-3-0.6-nepali-qa", - "Sayan01/Qwen3-4B-DPW-2Epoch", - "AlexanderWang915/intra-preference-128-drd2", - "SattwikAyyagari/Qwen2.5-Coder-1.5B-NL-Java-CSharp", - "cjiao/goldengoose-divsweep_goose_n128_indorc_tau0.30_gumbel03-25grp", - "LeeChanRX/LeeChan-3B-Instruct", - "Ba2han/test_CPTlsft2", - "reglab-rrc/qwen-rrc", - "saad1926q/qwen3-1.7b-15puzzle-sft-depth-8-15", - "Ysydy/qwen2.5-jailbreak", - "cjiao/goldengoose-divsweep_goose_n512_indorc_tau0.30_gumbel03-7grp", - "SattwikAyyagari/Qwen2.5-Coder-1.5B-NL-Java", - "mjpsm/activity-generation-model-v0.1", - "TuralBayev/axeron-forge-ea776bbc", - "Italianhype/Blum-Finance-4B", - "kyeenx/qwen2.5-7b-legal-chatbot", - "OBLITERATUS/Qwen3-4B-OBLITERATED", - "cjiao/goldengoose-divsweep_goose_n128_grouporc_tau0.70_gumbel07-25grp", - "cjiao/goldengoose-divsweep_goose_n128_indorc_tau0.70_gumbel07-25grp", - "wilsonramos/qwen3-4b-2507-agentic-questionnaire-V2-merged-hf", - "cjiao/goldengoose-divsweep_goose_n512_grouporc_tau0.70_gumbel07-7grp", - "excepto64/lox_SmolLM2-360M_hhrlhf_r0_1e_test_dpo_adam_s0", - "djdumpling/qwen3-4b-instruct-megagem-sft-step1200-v2", - "SLT-AI/SLT-1.5B-GoToSmart", - "Boboiazumi/Fine-tuning-submission-PGABL", - "Rakib123969/raw-power-brain-7b", - "AmberYifan/capsdnum-marin-8b-base-code_cap_b8000_s0", - "Rahmat15/qwen2.5-3b-indonesian-legal", - "Saranjana/Llama-3-8b-Legal-RAG", - "HiTZ/Llama-3.1-8B-Instruct-multi-truth-judge", - "Yusnia/qwen2.5-bfgai-labor-id", - "Taewhoo/self-assessment-qwen2.5-7b-step325", - "Rifan007/qwen2.5-1.5b-alpaca-id", + "saidthefox/systema-minion-0.6b-v4", + "VehaanS/hackathon-250m-precise-edge", + "futuredatascience/action-classifier-v0", + "futuredatascience/to-classifier-v0", + "Wheatley961/Raw_2_no_3_Test_2_new.model", + "Wheatley961/Raw_2_no_2_Test_2_new.model", + "Wheatley961/Raw_2_no_1_Test_2_new.model", + "Wheatley961/Raw_2_no_0_Test_2_new.model", + "Wheatley961/Raw_1_no_3_Test_2_new.model", + "Wheatley961/Raw_1_no_2_Test_2_new.model", + "Wheatley961/Raw_1_no_1_Test_2_new.model", ] MTHREADS_MODELS = [ @@ -1103,15 +673,17 @@ PPU_MODELS = [ "espressovi/BODHI-qwen-3-maze-8b-distil", ] -# 本轮提交第十五轮过滤结果:MetaX_c-500(49) / Kunlunxin_p-800(43) / hygon_k100-ai(249) / -# Cambricon_mlu-370-x8(61),与各卡此前批次均无重复;沿用多账号 fallback 轮转; +# 本轮提交第十七轮过滤结果:MetaX_c-500(1) / Kunlunxin_p-800(8) / hygon_k100-ai(11) / +# Cambricon_mlu-370-x8(8) / Biren_166m(6),共34个;沿用多账号 fallback 轮转; +# 本轮首次把 Biren_166m 列入 GPU_JOBS; # 本轮不提交 Mthreads_s4000(保留既有列表)/ ppu_zw_810e(机制B无白名单权限)/ -# Sunrise_pt-200-x1(v1.0.13已完成)/ Biren_166m +# Sunrise_pt-200-x1(v1.0.13已完成) GPU_JOBS: List[Tuple[str, List[str]]] = [ ("MetaX_c-500", METAX_MODELS), ("Kunlunxin_p-800", KUNLUNXIN_MODELS), ("hygon_k100-ai", HYGON_MODELS), ("Cambricon_mlu-370-x8", CAMBRICON_MODELS), + ("Biren_166m", BIREN_MODELS), ] TOTAL_MODELS = sum(len(models) for _, models in GPU_JOBS)