add tunning files for QWEN-3-NEXT (#10794)

This commit is contained in:
Yiakwy
2025-09-24 03:46:30 +08:00
committed by GitHub
parent 23632d350c
commit 984730b732
3 changed files with 297 additions and 1 deletions

View File

@@ -411,7 +411,11 @@ def main(args: argparse.Namespace):
topk = config.num_experts_per_tok
intermediate_size = config.intermediate_size
shard_intermediate_size = 2 * intermediate_size // args.tp_size
elif config.architectures[0] in ["Qwen2MoeForCausalLM", "Qwen3MoeForCausalLM"]:
elif config.architectures[0] in [
"Qwen2MoeForCausalLM",
"Qwen3MoeForCausalLM",
"Qwen3NextForCausalLM",
]:
E = config.num_experts
topk = config.num_experts_per_tok
intermediate_size = config.moe_intermediate_size