初始化项目,由ModelHub XC社区提供模型
Model: ayh015/myLightningOPD Source: Original Platform
This commit is contained in:
88
slime/backends/megatron_utils/megatron_to_hf/__init__.py
Normal file
88
slime/backends/megatron_utils/megatron_to_hf/__init__.py
Normal file
@@ -0,0 +1,88 @@
|
||||
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
from .deepseekv3 import convert_deepseekv3_to_hf
|
||||
from .glm4 import convert_glm4_to_hf
|
||||
from .glm4moe import convert_glm4moe_to_hf
|
||||
from .llama import convert_llama_to_hf
|
||||
from .mimo import convert_mimo_to_hf
|
||||
from .processors.padding_remover import remove_padding
|
||||
from .processors.quantizer import quantize_params
|
||||
from .qwen2 import convert_qwen2_to_hf
|
||||
from .qwen3_next import convert_qwen3_next_to_hf
|
||||
from .qwen3moe import convert_qwen3moe_to_hf
|
||||
|
||||
|
||||
# TODO unify w/ `convert_to_hf`
|
||||
def postprocess_hf_param(args, megatron_param_name, hf_param_name, param):
|
||||
param = remove_padding(megatron_param_name, param, args.vocab_size)
|
||||
# TODO support quant
|
||||
return param
|
||||
|
||||
|
||||
# TODO optimize code details
|
||||
def convert_to_hf(args, model_name, name, param, quantization_config=None):
|
||||
param = remove_padding(name, param, args.vocab_size)
|
||||
|
||||
converted_named_tensors = _convert_to_hf_core(args, model_name, name, param)
|
||||
|
||||
if not quantization_config:
|
||||
return converted_named_tensors
|
||||
|
||||
return quantize_params(args, name, converted_named_tensors, quantization_config)
|
||||
|
||||
|
||||
# TODO optimize
|
||||
_cached_tensors = {}
|
||||
|
||||
|
||||
# TODO optimize code details
|
||||
def _convert_to_hf_core(args, model_name, name, param):
|
||||
if "glm4moe" in model_name:
|
||||
converted_named_tensors = convert_glm4moe_to_hf(args, name, param)
|
||||
elif "glm4" in model_name:
|
||||
converted_named_tensors = convert_glm4_to_hf(args, name, param)
|
||||
elif "qwen3moe" in model_name:
|
||||
converted_named_tensors = convert_qwen3moe_to_hf(args, name, param)
|
||||
elif "qwen3next" in model_name:
|
||||
converted_named_tensors = convert_qwen3_next_to_hf(args, name, param)
|
||||
elif "qwen2" in model_name or "qwen3" in model_name:
|
||||
converted_named_tensors = convert_qwen2_to_hf(args, name, param)
|
||||
elif "deepseekv3" in model_name:
|
||||
converted_named_tensors = convert_deepseekv3_to_hf(args, name, param)
|
||||
|
||||
elif "llama" in model_name:
|
||||
converted_named_tensors = convert_llama_to_hf(args, name, param)
|
||||
elif "mimo" in model_name:
|
||||
converted_named_tensors = convert_mimo_to_hf(args, name, param)
|
||||
else:
|
||||
raise ValueError(f"Unsupported model: {model_name}")
|
||||
|
||||
# to compatible with sglang implementation
|
||||
if args.q_lora_rank is not None:
|
||||
old_converted_named_tensors = converted_named_tensors
|
||||
converted_named_tensors = []
|
||||
for converted_name, converted_param in old_converted_named_tensors:
|
||||
if "q_a_proj" in converted_name:
|
||||
pair_name = converted_name.replace("q_a_proj", "kv_a_proj_with_mqa")
|
||||
if pair_name in _cached_tensors:
|
||||
converted_named_tensors += [
|
||||
(converted_name, converted_param),
|
||||
(pair_name, _cached_tensors[pair_name]),
|
||||
]
|
||||
del _cached_tensors[pair_name]
|
||||
else:
|
||||
_cached_tensors[converted_name] = converted_param
|
||||
elif "kv_a_proj_with_mqa" in converted_name:
|
||||
pair_name = converted_name.replace("kv_a_proj_with_mqa", "q_a_proj")
|
||||
if pair_name in _cached_tensors:
|
||||
converted_named_tensors += [
|
||||
(converted_name, converted_param),
|
||||
(pair_name, _cached_tensors[pair_name]),
|
||||
]
|
||||
del _cached_tensors[pair_name]
|
||||
else:
|
||||
_cached_tensors[converted_name] = converted_param
|
||||
else:
|
||||
converted_named_tensors.append((converted_name, converted_param))
|
||||
return converted_named_tensors
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
132
slime/backends/megatron_utils/megatron_to_hf/deepseekv3.py
Normal file
132
slime/backends/megatron_utils/megatron_to_hf/deepseekv3.py
Normal file
@@ -0,0 +1,132 @@
|
||||
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
import re
|
||||
|
||||
import torch
|
||||
|
||||
|
||||
def convert_deepseekv3_to_hf(args, name, param):
|
||||
if name == "module.module.embedding.word_embeddings.weight":
|
||||
return [("model.embed_tokens.weight", param)]
|
||||
if name == "module.module.output_layer.weight":
|
||||
return [("lm_head.weight", param)]
|
||||
if name == "module.module.decoder.final_layernorm.weight":
|
||||
return [("model.norm.weight", param)]
|
||||
|
||||
try:
|
||||
head_dim = args.kv_channels if args.kv_channels is not None else args.hidden_size // args.num_attention_heads
|
||||
except AttributeError:
|
||||
head_dim = args.hidden_size // args.num_attention_heads
|
||||
value_num_per_group = args.num_attention_heads // args.num_query_groups
|
||||
|
||||
decoder_layers_pattern = r"module\.module\.decoder\.layers\.(\d+)\.(.+)"
|
||||
match = re.match(decoder_layers_pattern, name)
|
||||
if match:
|
||||
layer_idx, rest = match.groups()
|
||||
|
||||
# experts
|
||||
expert_pattern = r"mlp.experts\.(.+)\.weight(\d+)"
|
||||
match = re.match(expert_pattern, rest)
|
||||
if match:
|
||||
rest, expert_idx = match.groups()
|
||||
if rest == "linear_fc1":
|
||||
gate_weight, up_weight = param.chunk(2, dim=0)
|
||||
outputs = [
|
||||
(f"model.layers.{layer_idx}.mlp.experts.{expert_idx}.gate_proj.weight", gate_weight),
|
||||
(f"model.layers.{layer_idx}.mlp.experts.{expert_idx}.up_proj.weight", up_weight),
|
||||
]
|
||||
return outputs
|
||||
elif rest == "linear_fc2":
|
||||
outputs = [
|
||||
(f"model.layers.{layer_idx}.mlp.experts.{expert_idx}.down_proj.weight", param),
|
||||
]
|
||||
return outputs
|
||||
else:
|
||||
raise ValueError(f"Unknown expert parameter name: {name}")
|
||||
|
||||
# shared expert
|
||||
shared_expert_pattern = r"mlp.shared_experts\.(.+)"
|
||||
match = re.match(shared_expert_pattern, rest)
|
||||
if match:
|
||||
rest = match.groups()[0]
|
||||
if rest == "linear_fc1.weight":
|
||||
gate_weight, up_weight = param.chunk(2, dim=0)
|
||||
return [
|
||||
(f"model.layers.{layer_idx}.mlp.shared_experts.gate_proj.weight", gate_weight),
|
||||
(f"model.layers.{layer_idx}.mlp.shared_experts.up_proj.weight", up_weight),
|
||||
]
|
||||
elif rest == "linear_fc2.weight":
|
||||
return [(f"model.layers.{layer_idx}.mlp.shared_experts.down_proj.weight", param)]
|
||||
else:
|
||||
raise ValueError(f"Unknown shared expert parameter name: {name}")
|
||||
|
||||
if rest == "self_attention.linear_proj.weight":
|
||||
return [(f"model.layers.{layer_idx}.self_attn.o_proj.weight", param)]
|
||||
elif rest == "self_attention.linear_q_proj.weight":
|
||||
return [(f"model.layers.{layer_idx}.self_attn.q_proj.weight", param)]
|
||||
elif rest == "self_attention.linear_q_down_proj.weight":
|
||||
return [(f"model.layers.{layer_idx}.self_attn.q_a_proj.weight", param)]
|
||||
elif rest == "self_attention.linear_q_up_proj.layer_norm_weight":
|
||||
return [(f"model.layers.{layer_idx}.self_attn.q_a_layernorm.weight", param)]
|
||||
elif rest == "self_attention.linear_q_up_proj.weight":
|
||||
return [(f"model.layers.{layer_idx}.self_attn.q_b_proj.weight", param)]
|
||||
elif rest == "self_attention.linear_qkv.bias":
|
||||
param = param.view(args.num_query_groups, -1)
|
||||
q_bias, k_bias, v_bias = torch.split(
|
||||
param,
|
||||
split_size_or_sections=[value_num_per_group * head_dim, head_dim, head_dim],
|
||||
dim=1,
|
||||
)
|
||||
q_bias = q_bias.contiguous().flatten()
|
||||
k_bias = k_bias.contiguous().flatten()
|
||||
v_bias = v_bias.contiguous().flatten()
|
||||
return [
|
||||
(f"model.layers.{layer_idx}.self_attn.q_proj.bias", q_bias),
|
||||
(f"model.layers.{layer_idx}.self_attn.k_proj.bias", k_bias),
|
||||
(f"model.layers.{layer_idx}.self_attn.v_proj.bias", v_bias),
|
||||
]
|
||||
elif rest == "mlp.linear_fc1.weight":
|
||||
gate_weight, up_weight = param.chunk(2, dim=0)
|
||||
return [
|
||||
(f"model.layers.{layer_idx}.mlp.gate_proj.weight", gate_weight),
|
||||
(f"model.layers.{layer_idx}.mlp.up_proj.weight", up_weight),
|
||||
]
|
||||
elif rest == "mlp.linear_fc2.weight":
|
||||
return [(f"model.layers.{layer_idx}.mlp.down_proj.weight", param)]
|
||||
elif rest == "self_attention.linear_qkv.layer_norm_weight" or rest == "input_layernorm.weight":
|
||||
return [(f"model.layers.{layer_idx}.input_layernorm.weight", param)]
|
||||
elif rest == "mlp.linear_fc1.layer_norm_weight":
|
||||
return [(f"model.layers.{layer_idx}.post_attention_layernorm.weight", param)]
|
||||
elif rest == "self_attention.linear_kv_down_proj.weight":
|
||||
return [(f"model.layers.{layer_idx}.self_attn.kv_a_proj_with_mqa.weight", param)]
|
||||
elif rest == "self_attention.linear_kv_up_proj.layer_norm_weight":
|
||||
return [(f"model.layers.{layer_idx}.self_attn.kv_a_layernorm.weight", param)]
|
||||
elif rest == "self_attention.linear_kv_up_proj.weight":
|
||||
return [(f"model.layers.{layer_idx}.self_attn.kv_b_proj.weight", param)]
|
||||
elif rest == "pre_mlp_layernorm.weight":
|
||||
return [(f"model.layers.{layer_idx}.post_attention_layernorm.weight", param)]
|
||||
elif rest == "mlp.router.weight":
|
||||
return [(f"model.layers.{layer_idx}.mlp.gate.weight", param)]
|
||||
elif rest == "mlp.router.expert_bias":
|
||||
return [(f"model.layers.{layer_idx}.mlp.gate.e_score_correction_bias", param)]
|
||||
|
||||
mtp_layer_pattern = r"module\.module\.mtp\.layers\.(\d+)\.(.+)"
|
||||
match = re.match(mtp_layer_pattern, name)
|
||||
if match:
|
||||
layer_idx, rest = match.groups()
|
||||
layer_idx = int(layer_idx) + args.num_layers
|
||||
if rest == "eh_proj.weight":
|
||||
return [(f"model.layers.{layer_idx}.eh_proj.weight", param)]
|
||||
elif rest == "enorm.weight":
|
||||
return [(f"model.layers.{layer_idx}.enorm.weight", param)]
|
||||
elif rest == "hnorm.weight":
|
||||
return [(f"model.layers.{layer_idx}.hnorm.weight", param)]
|
||||
elif rest == "final_layernorm.weight":
|
||||
return [(f"model.layers.{layer_idx}.shared_head.norm.weight", param)]
|
||||
else:
|
||||
name = f"module.module.decoder.layers.{layer_idx}.{rest}"
|
||||
name = name.replace("transformer_layer.", "")
|
||||
return convert_deepseekv3_to_hf(args, name, param)
|
||||
|
||||
raise ValueError(f"Unknown parameter name: {name}")
|
||||
78
slime/backends/megatron_utils/megatron_to_hf/glm4.py
Normal file
78
slime/backends/megatron_utils/megatron_to_hf/glm4.py
Normal file
@@ -0,0 +1,78 @@
|
||||
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
import re
|
||||
import torch
|
||||
|
||||
|
||||
def convert_glm4_to_hf(args, name, param):
|
||||
if name == "module.module.embedding.word_embeddings.weight":
|
||||
return [("model.embed_tokens.weight", param)]
|
||||
if name == "module.module.output_layer.weight":
|
||||
return [("lm_head.weight", param)]
|
||||
if name == "module.module.decoder.final_layernorm.weight":
|
||||
return [("model.norm.weight", param)]
|
||||
|
||||
try:
|
||||
head_dim = args.kv_channels if args.kv_channels is not None else args.hidden_size // args.num_attention_heads
|
||||
except AttributeError:
|
||||
head_dim = args.hidden_size // args.num_attention_heads
|
||||
value_num_per_group = args.num_attention_heads // args.num_query_groups
|
||||
|
||||
decoder_layers_pattern = r"module\.module\.decoder\.layers\.(\d+)\.(.+)"
|
||||
match = re.match(decoder_layers_pattern, name)
|
||||
if match:
|
||||
layer_idx, rest = match.groups()
|
||||
if rest == "self_attention.linear_proj.weight":
|
||||
return [(f"model.layers.{layer_idx}.self_attn.o_proj.weight", param)]
|
||||
elif rest == "self_attention.linear_qkv.weight":
|
||||
|
||||
param = param.view(args.num_query_groups, -1, head_dim, args.hidden_size)
|
||||
q_param, k_param, v_param = torch.split(param, split_size_or_sections=[value_num_per_group, 1, 1], dim=1)
|
||||
q_param = q_param.reshape(-1, args.hidden_size)
|
||||
k_param = k_param.reshape(-1, args.hidden_size)
|
||||
v_param = v_param.reshape(-1, args.hidden_size)
|
||||
return [
|
||||
(f"model.layers.{layer_idx}.self_attn.q_proj.weight", q_param),
|
||||
(f"model.layers.{layer_idx}.self_attn.k_proj.weight", k_param),
|
||||
(f"model.layers.{layer_idx}.self_attn.v_proj.weight", v_param),
|
||||
]
|
||||
elif rest == "self_attention.linear_qkv.bias":
|
||||
param = param.view(args.num_query_groups, -1)
|
||||
q_bias, k_bias, v_bias = torch.split(
|
||||
param,
|
||||
split_size_or_sections=[value_num_per_group * head_dim, head_dim, head_dim],
|
||||
dim=1,
|
||||
)
|
||||
q_bias = q_bias.contiguous().flatten()
|
||||
k_bias = k_bias.contiguous().flatten()
|
||||
v_bias = v_bias.contiguous().flatten()
|
||||
return [
|
||||
(f"model.layers.{layer_idx}.self_attn.q_proj.bias", q_bias),
|
||||
(f"model.layers.{layer_idx}.self_attn.k_proj.bias", k_bias),
|
||||
(f"model.layers.{layer_idx}.self_attn.v_proj.bias", v_bias),
|
||||
]
|
||||
elif rest == "mlp.linear_fc1.weight":
|
||||
return [
|
||||
(f"model.layers.{layer_idx}.mlp.gate_up_proj.weight", param),
|
||||
]
|
||||
elif rest == "mlp.linear_fc2.weight":
|
||||
return [(f"model.layers.{layer_idx}.mlp.down_proj.weight", param)]
|
||||
elif rest == "self_attention.linear_qkv.layer_norm_weight":
|
||||
return [(f"model.layers.{layer_idx}.input_layernorm.weight", param)]
|
||||
elif rest == "mlp.linear_fc1.layer_norm_weight":
|
||||
return [(f"model.layers.{layer_idx}.post_attention_layernorm.weight", param)]
|
||||
|
||||
# qk norm
|
||||
elif rest == "self_attention.q_layernorm.weight":
|
||||
return [(f"model.layers.{layer_idx}.self_attn.q_norm.weight", param)]
|
||||
elif rest == "self_attention.k_layernorm.weight":
|
||||
return [(f"model.layers.{layer_idx}.self_attn.k_norm.weight", param)]
|
||||
|
||||
# sandwitch norm
|
||||
elif rest == "post_self_attn_layernorm.weight":
|
||||
return [(f"model.layers.{layer_idx}.post_self_attn_layernorm.weight", param)]
|
||||
elif rest == "post_mlp_layernorm.weight":
|
||||
return [(f"model.layers.{layer_idx}.post_mlp_layernorm.weight", param)]
|
||||
|
||||
raise ValueError(f"Unknown parameter name: {name}")
|
||||
142
slime/backends/megatron_utils/megatron_to_hf/glm4moe.py
Normal file
142
slime/backends/megatron_utils/megatron_to_hf/glm4moe.py
Normal file
@@ -0,0 +1,142 @@
|
||||
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
import re
|
||||
|
||||
import torch
|
||||
|
||||
|
||||
def convert_glm4moe_to_hf(args, name, param):
|
||||
if name == "module.module.embedding.word_embeddings.weight":
|
||||
return [("model.embed_tokens.weight", param)]
|
||||
if name == "module.module.output_layer.weight":
|
||||
return [("lm_head.weight", param)]
|
||||
if name == "module.module.decoder.final_layernorm.weight":
|
||||
return [("model.norm.weight", param)]
|
||||
|
||||
try:
|
||||
head_dim = args.kv_channels if args.kv_channels is not None else args.hidden_size // args.num_attention_heads
|
||||
except AttributeError:
|
||||
head_dim = args.hidden_size // args.num_attention_heads
|
||||
value_num_per_group = args.num_attention_heads // args.num_query_groups
|
||||
|
||||
decoder_layers_pattern = r"module\.module\.decoder\.layers\.(\d+)\.(.+)"
|
||||
match = re.match(decoder_layers_pattern, name)
|
||||
if match:
|
||||
layer_idx, rest = match.groups()
|
||||
|
||||
# experts
|
||||
expert_pattern = r"mlp.experts\.(.+)\.weight(\d+)"
|
||||
match = re.match(expert_pattern, rest)
|
||||
if match:
|
||||
rest, expert_idx = match.groups()
|
||||
if rest == "linear_fc1":
|
||||
gate_weight, up_weight = param.chunk(2, dim=0)
|
||||
outputs = [
|
||||
(f"model.layers.{layer_idx}.mlp.experts.{expert_idx}.gate_proj.weight", gate_weight),
|
||||
(f"model.layers.{layer_idx}.mlp.experts.{expert_idx}.up_proj.weight", up_weight),
|
||||
]
|
||||
return outputs
|
||||
elif rest == "linear_fc2":
|
||||
outputs = [
|
||||
(f"model.layers.{layer_idx}.mlp.experts.{expert_idx}.down_proj.weight", param),
|
||||
]
|
||||
return outputs
|
||||
else:
|
||||
raise ValueError(f"Unknown expert parameter name: {name}")
|
||||
|
||||
# shared expert
|
||||
shared_expert_pattern = r"mlp.shared_experts\.(.+)"
|
||||
match = re.match(shared_expert_pattern, rest)
|
||||
if match:
|
||||
rest = match.groups()[0]
|
||||
if rest == "linear_fc1.weight":
|
||||
gate_weight, up_weight = param.chunk(2, dim=0)
|
||||
return [
|
||||
(f"model.layers.{layer_idx}.mlp.shared_experts.gate_proj.weight", gate_weight),
|
||||
(f"model.layers.{layer_idx}.mlp.shared_experts.up_proj.weight", up_weight),
|
||||
]
|
||||
elif rest == "linear_fc2.weight":
|
||||
return [
|
||||
(f"model.layers.{layer_idx}.mlp.shared_experts.down_proj.weight", param),
|
||||
]
|
||||
else:
|
||||
raise ValueError(f"Unknown shared expert parameter name: {name}")
|
||||
|
||||
if rest == "self_attention.linear_proj.weight":
|
||||
return [(f"model.layers.{layer_idx}.self_attn.o_proj.weight", param)]
|
||||
elif rest == "self_attention.linear_qkv.weight":
|
||||
|
||||
param = param.view(args.num_query_groups, -1, head_dim, args.hidden_size)
|
||||
q_param, k_param, v_param = torch.split(param, split_size_or_sections=[value_num_per_group, 1, 1], dim=1)
|
||||
q_param = q_param.reshape(-1, args.hidden_size)
|
||||
k_param = k_param.reshape(-1, args.hidden_size)
|
||||
v_param = v_param.reshape(-1, args.hidden_size)
|
||||
return [
|
||||
(f"model.layers.{layer_idx}.self_attn.q_proj.weight", q_param),
|
||||
(f"model.layers.{layer_idx}.self_attn.k_proj.weight", k_param),
|
||||
(f"model.layers.{layer_idx}.self_attn.v_proj.weight", v_param),
|
||||
]
|
||||
elif rest == "self_attention.linear_qkv.bias":
|
||||
param = param.view(args.num_query_groups, -1)
|
||||
q_bias, k_bias, v_bias = torch.split(
|
||||
param,
|
||||
split_size_or_sections=[value_num_per_group * head_dim, head_dim, head_dim],
|
||||
dim=1,
|
||||
)
|
||||
q_bias = q_bias.contiguous().flatten()
|
||||
k_bias = k_bias.contiguous().flatten()
|
||||
v_bias = v_bias.contiguous().flatten()
|
||||
return [
|
||||
(f"model.layers.{layer_idx}.self_attn.q_proj.bias", q_bias),
|
||||
(f"model.layers.{layer_idx}.self_attn.k_proj.bias", k_bias),
|
||||
(f"model.layers.{layer_idx}.self_attn.v_proj.bias", v_bias),
|
||||
]
|
||||
elif rest == "mlp.linear_fc1.weight":
|
||||
gate_weight, up_weight = param.chunk(2, dim=0)
|
||||
return [
|
||||
(f"model.layers.{layer_idx}.mlp.gate_proj.weight", gate_weight),
|
||||
(f"model.layers.{layer_idx}.mlp.up_proj.weight", up_weight),
|
||||
]
|
||||
elif rest == "mlp.linear_fc2.weight":
|
||||
return [(f"model.layers.{layer_idx}.mlp.down_proj.weight", param)]
|
||||
elif rest == "self_attention.linear_qkv.layer_norm_weight":
|
||||
return [(f"model.layers.{layer_idx}.input_layernorm.weight", param)]
|
||||
elif rest == "mlp.linear_fc1.layer_norm_weight":
|
||||
return [(f"model.layers.{layer_idx}.post_attention_layernorm.weight", param)]
|
||||
elif rest == "post_self_attn_layernorm.weight":
|
||||
return [(f"model.layers.{layer_idx}.post_self_attn_layernorm.weight", param)]
|
||||
elif rest == "post_mlp_layernorm.weight":
|
||||
return [(f"model.layers.{layer_idx}.post_mlp_layernorm.weight", param)]
|
||||
elif rest == "pre_mlp_layernorm.weight":
|
||||
return [(f"model.layers.{layer_idx}.post_attention_layernorm.weight", param)]
|
||||
elif rest == "mlp.router.weight":
|
||||
return [(f"model.layers.{layer_idx}.mlp.gate.weight", param)]
|
||||
elif rest == "mlp.router.expert_bias":
|
||||
return [(f"model.layers.{layer_idx}.mlp.gate.e_score_correction_bias", param)]
|
||||
|
||||
# qk norm
|
||||
elif rest == "self_attention.q_layernorm.weight":
|
||||
return [(f"model.layers.{layer_idx}.self_attn.q_norm.weight", param)]
|
||||
elif rest == "self_attention.k_layernorm.weight":
|
||||
return [(f"model.layers.{layer_idx}.self_attn.k_norm.weight", param)]
|
||||
|
||||
mtp_layer_pattern = r"module\.module\.mtp\.layers\.(\d+)\.(.+)"
|
||||
match = re.match(mtp_layer_pattern, name)
|
||||
if match:
|
||||
layer_idx, rest = match.groups()
|
||||
layer_idx = int(layer_idx) + args.num_layers
|
||||
if rest == "eh_proj.weight":
|
||||
return [(f"model.layers.{layer_idx}.eh_proj.weight", param)]
|
||||
elif rest == "enorm.weight":
|
||||
return [(f"model.layers.{layer_idx}.enorm.weight", param)]
|
||||
elif rest == "hnorm.weight":
|
||||
return [(f"model.layers.{layer_idx}.hnorm.weight", param)]
|
||||
elif rest == "final_layernorm.weight":
|
||||
return [(f"model.layers.{layer_idx}.shared_head.norm.weight", param)]
|
||||
else:
|
||||
name = f"module.module.decoder.layers.{layer_idx}.{rest}"
|
||||
name = name.replace("transformer_layer.", "")
|
||||
return convert_glm4moe_to_hf(args, name, param)
|
||||
|
||||
raise ValueError(f"Unknown parameter name: {name}")
|
||||
56
slime/backends/megatron_utils/megatron_to_hf/llama.py
Normal file
56
slime/backends/megatron_utils/megatron_to_hf/llama.py
Normal file
@@ -0,0 +1,56 @@
|
||||
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
import re
|
||||
import torch
|
||||
|
||||
|
||||
def convert_llama_to_hf(args, name, param):
|
||||
if name == "module.module.embedding.word_embeddings.weight":
|
||||
return [("model.embed_tokens.weight", param)]
|
||||
if name == "module.module.output_layer.weight":
|
||||
return [("lm_head.weight", param)]
|
||||
if name == "module.module.decoder.final_layernorm.weight":
|
||||
return [("model.norm.weight", param)]
|
||||
|
||||
try:
|
||||
head_dim = args.kv_channels if args.kv_channels is not None else args.hidden_size // args.num_attention_heads
|
||||
except AttributeError:
|
||||
head_dim = args.hidden_size // args.num_attention_heads
|
||||
value_num_per_group = args.num_attention_heads // args.num_query_groups
|
||||
|
||||
decoder_layers_pattern = r"module\.module\.decoder\.layers\.(\d+)\.(.+)"
|
||||
match = re.match(decoder_layers_pattern, name)
|
||||
if match:
|
||||
layer_idx, rest = match.groups()
|
||||
if rest == "self_attention.linear_proj.weight":
|
||||
return [(f"model.layers.{layer_idx}.self_attn.o_proj.weight", param)]
|
||||
elif rest == "self_attention.linear_qkv.weight":
|
||||
# Split QKV weight for Llama
|
||||
param = param.view(args.num_query_groups, -1, head_dim, args.hidden_size)
|
||||
q_param, k_param, v_param = torch.split(param, split_size_or_sections=[value_num_per_group, 1, 1], dim=1)
|
||||
q_param = q_param.reshape(-1, args.hidden_size)
|
||||
k_param = k_param.reshape(-1, args.hidden_size)
|
||||
v_param = v_param.reshape(-1, args.hidden_size)
|
||||
return [
|
||||
(f"model.layers.{layer_idx}.self_attn.q_proj.weight", q_param),
|
||||
(f"model.layers.{layer_idx}.self_attn.k_proj.weight", k_param),
|
||||
(f"model.layers.{layer_idx}.self_attn.v_proj.weight", v_param),
|
||||
]
|
||||
elif rest == "mlp.linear_fc1.weight":
|
||||
# Split gate and up projections for SwiGLU
|
||||
gate_weight, up_weight = param.chunk(2, dim=0)
|
||||
return [
|
||||
(f"model.layers.{layer_idx}.mlp.gate_proj.weight", gate_weight),
|
||||
(f"model.layers.{layer_idx}.mlp.up_proj.weight", up_weight),
|
||||
]
|
||||
elif rest == "mlp.linear_fc2.weight":
|
||||
return [(f"model.layers.{layer_idx}.mlp.down_proj.weight", param)]
|
||||
elif rest == "self_attention.linear_qkv.layer_norm_weight":
|
||||
return [(f"model.layers.{layer_idx}.input_layernorm.weight", param)]
|
||||
elif rest == "mlp.linear_fc1.layer_norm_weight":
|
||||
return [(f"model.layers.{layer_idx}.post_attention_layernorm.weight", param)]
|
||||
elif rest == "pre_mlp_layernorm.weight":
|
||||
return [(f"model.layers.{layer_idx}.post_attention_layernorm.weight", param)]
|
||||
|
||||
raise ValueError(f"Unknown parameter name: {name}")
|
||||
78
slime/backends/megatron_utils/megatron_to_hf/mimo.py
Normal file
78
slime/backends/megatron_utils/megatron_to_hf/mimo.py
Normal file
@@ -0,0 +1,78 @@
|
||||
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
import re
|
||||
import torch
|
||||
from .qwen2 import convert_qwen2_to_hf
|
||||
|
||||
|
||||
def convert_mimo_to_hf(args, name, param):
|
||||
"""
|
||||
Convert MiMo model parameters from Megatron to HuggingFace format.
|
||||
|
||||
MiMo extends Qwen2 with MTP (Multi-Token Prediction) layers.
|
||||
"""
|
||||
|
||||
if "mtp" in name:
|
||||
return convert_mimo_mtp_param(args, name, param)
|
||||
|
||||
return convert_qwen2_to_hf(args, name, param)
|
||||
|
||||
|
||||
def convert_mimo_mtp_param(args, name, param):
|
||||
"""
|
||||
Convert MTP layer parameters from Megatron to HuggingFace format.
|
||||
|
||||
MTP layers in MiMo contain:
|
||||
- LayerNorms (token_layernorm, hidden_layernorm, final_layernorm)
|
||||
- Input projection (input_proj)
|
||||
- Self attention (reuses Qwen2 attention structure)
|
||||
- MLP (reuses Qwen2 MLP structure)
|
||||
|
||||
Based on MimoBridge._convert_mtp_param logic (reverse mapping)
|
||||
"""
|
||||
mtp_pattern = r"module\.module\.mtp\.layers\.(\d+)\.(.+)"
|
||||
match = re.match(mtp_pattern, name)
|
||||
|
||||
if not match:
|
||||
raise ValueError(f"Invalid MTP parameter name: {name}")
|
||||
|
||||
layer_idx, component = match.groups()
|
||||
|
||||
# Direct mappings for MTP-specific components (Megatron -> HF)
|
||||
# Based on MimoBridge direct_name_mapping (reversed)
|
||||
direct_mappings = {
|
||||
"enorm.weight": f"model.mtp_layers.{layer_idx}.token_layernorm.weight",
|
||||
"hnorm.weight": f"model.mtp_layers.{layer_idx}.hidden_layernorm.weight",
|
||||
"eh_proj.weight": f"model.mtp_layers.{layer_idx}.input_proj.weight",
|
||||
"final_layernorm.weight": f"model.mtp_layers.{layer_idx}.final_layernorm.weight",
|
||||
}
|
||||
if component == "eh_proj.weight":
|
||||
first_half, second_half = param.chunk(2, dim=1)
|
||||
param = torch.cat([second_half, first_half], dim=1)
|
||||
|
||||
# Check direct mappings first
|
||||
if component in direct_mappings:
|
||||
return [(direct_mappings[component], param)]
|
||||
|
||||
# Handle transformer_layer components
|
||||
if component.startswith("transformer_layer."):
|
||||
# Remove "transformer_layer." prefix
|
||||
transformer_component = component[len("transformer_layer.") :]
|
||||
|
||||
# Create proxy name for reusing existing Qwen2 conversion functions
|
||||
proxy_name = f"module.module.decoder.layers.{layer_idx}.{transformer_component}"
|
||||
|
||||
# Use existing convert_qwen2_to_hf function for transformer components
|
||||
results = convert_qwen2_to_hf(args, proxy_name, param)
|
||||
|
||||
# Replace model.layers with mtp_layers in results
|
||||
converted_results = []
|
||||
for hf_name, hf_param in results:
|
||||
# Replace model.layers.{idx} with mtp_layers.{idx}
|
||||
hf_name = hf_name.replace(f"model.layers.{layer_idx}", f"model.mtp_layers.{layer_idx}")
|
||||
converted_results.append((hf_name, hf_param))
|
||||
|
||||
return converted_results
|
||||
|
||||
raise ValueError(f"Unknown MTP component: {component} in {name}")
|
||||
@@ -0,0 +1,3 @@
|
||||
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,15 @@
|
||||
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
import torch
|
||||
|
||||
from slime.backends.megatron_utils.misc_utils import strip_param_name_prefix
|
||||
|
||||
|
||||
def remove_padding(name: str, param: torch.Tensor, vocab_size: int) -> torch.Tensor:
|
||||
"""
|
||||
Remove vocab padding: param[:vocab_size] for embedding/output layers, else unchanged.
|
||||
"""
|
||||
if strip_param_name_prefix(name) in {"embedding.word_embeddings.weight", "output_layer.weight"}:
|
||||
return param[:vocab_size]
|
||||
return param
|
||||
@@ -0,0 +1,110 @@
|
||||
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
import re
|
||||
|
||||
import torch
|
||||
|
||||
from slime.utils.fp8_kernel import blockwise_cast_to_fp8_triton
|
||||
|
||||
from ...sglang import quant_weight_ue8m0, should_deepgemm_weight_requant_ue8m0, transform_scale_ue8m0
|
||||
|
||||
|
||||
def quantize_params(args, megatron_name, converted_named_params, quantization_config):
|
||||
if quantization_config is None:
|
||||
return converted_named_params
|
||||
assert quantization_config["quant_method"] == "fp8"
|
||||
assert quantization_config["fmt"] == "e4m3"
|
||||
assert quantization_config["activation_scheme"] == "dynamic"
|
||||
weight_block_size = quantization_config.get("weight_block_size", None)
|
||||
|
||||
decoder_layers_pattern = r"module\.module\.decoder\.layers\.(\d+)\.(.+)"
|
||||
match = re.match(decoder_layers_pattern, megatron_name)
|
||||
|
||||
if not match:
|
||||
# check mtp layers
|
||||
mtp_layer_pattern = r"module\.module\.mtp\.layers\.(\d+)\.(.+)"
|
||||
match = re.match(mtp_layer_pattern, megatron_name)
|
||||
if not match:
|
||||
return converted_named_params
|
||||
layer_idx, rest = match.groups()
|
||||
rest = rest.replace("transformer_layer.", "")
|
||||
else:
|
||||
layer_idx, rest = match.groups()
|
||||
|
||||
# experts
|
||||
expert_pattern = r"mlp.experts\.(.+)\.weight(\d+)"
|
||||
match = re.match(expert_pattern, rest)
|
||||
if match:
|
||||
rest, expert_idx = match.groups()
|
||||
if rest in [
|
||||
"linear_fc1",
|
||||
"linear_fc2",
|
||||
]:
|
||||
quantize_named_params = []
|
||||
for converted_name, param in converted_named_params:
|
||||
# skip bf16 weight_scale and input_scale
|
||||
# TODO: find a clearer way.
|
||||
if converted_name.endswith("_scale"):
|
||||
continue
|
||||
quantize_named_params.extend(_quantize_param(converted_name, param, weight_block_size))
|
||||
|
||||
return quantize_named_params
|
||||
|
||||
# shared expert
|
||||
shared_expert_pattern = r"mlp.shared_experts\.(.+)"
|
||||
match = re.match(shared_expert_pattern, rest)
|
||||
if match:
|
||||
rest = match.groups()[0]
|
||||
if rest in [
|
||||
"linear_fc1.weight",
|
||||
"linear_fc2.weight",
|
||||
]:
|
||||
quantize_named_params = []
|
||||
for converted_name, param in converted_named_params:
|
||||
quantize_named_params.extend(_quantize_param(converted_name, param, weight_block_size))
|
||||
|
||||
return quantize_named_params
|
||||
|
||||
if rest in [
|
||||
"self_attention.linear_proj.weight",
|
||||
"self_attention.linear_qkv.weight",
|
||||
"mlp.linear_fc1.weight",
|
||||
"mlp.linear_fc2.weight",
|
||||
# mla
|
||||
"self_attention.linear_q_proj.weight",
|
||||
"self_attention.linear_q_down_proj.weight",
|
||||
"self_attention.linear_q_up_proj.weight",
|
||||
"self_attention.linear_kv_down_proj.weight",
|
||||
"self_attention.linear_kv_up_proj.weight",
|
||||
]:
|
||||
quantize_named_params = []
|
||||
for converted_name, param in converted_named_params:
|
||||
quantize_named_params.extend(_quantize_param(converted_name, param, weight_block_size))
|
||||
|
||||
return quantize_named_params
|
||||
|
||||
# for other parameters, we just return the original converted_named_params
|
||||
return converted_named_params
|
||||
|
||||
|
||||
def _quantize_param(name, weight, weight_block_size):
|
||||
assert name.endswith(".weight"), f"Expected weight parameter, got {name}"
|
||||
FP8_MIN = torch.finfo(torch.float8_e4m3fn).min
|
||||
FP8_MAX = torch.finfo(torch.float8_e4m3fn).max
|
||||
if weight_block_size is not None:
|
||||
if should_deepgemm_weight_requant_ue8m0 and should_deepgemm_weight_requant_ue8m0(
|
||||
weight_block_size=weight_block_size
|
||||
):
|
||||
qweight, scale = quant_weight_ue8m0(weight, weight_block_size=weight_block_size)
|
||||
scale = transform_scale_ue8m0(scale, mn=qweight.shape[-2])
|
||||
else:
|
||||
qweight, scale = blockwise_cast_to_fp8_triton(weight, weight_block_size)
|
||||
scale_name = name.replace(".weight", ".weight_scale_inv")
|
||||
else:
|
||||
# per tensor quant
|
||||
scale = weight.abs().max().clamp(min=1e-12).to(torch.float32) / FP8_MAX
|
||||
qweight = (weight / scale).clamp(min=FP8_MIN, max=FP8_MAX).to(torch.float8_e4m3fn)
|
||||
scale = scale.view(1)
|
||||
scale_name = name.replace(".weight", ".weight_scale")
|
||||
return [(name, qweight), (scale_name, scale)]
|
||||
74
slime/backends/megatron_utils/megatron_to_hf/qwen2.py
Normal file
74
slime/backends/megatron_utils/megatron_to_hf/qwen2.py
Normal file
@@ -0,0 +1,74 @@
|
||||
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
import re
|
||||
import torch
|
||||
|
||||
|
||||
def convert_qwen2_to_hf(args, name, param):
|
||||
if name == "module.module.embedding.word_embeddings.weight":
|
||||
return [("model.embed_tokens.weight", param)]
|
||||
if name == "module.module.output_layer.weight":
|
||||
return [("lm_head.weight", param)]
|
||||
if name == "module.module.decoder.final_layernorm.weight":
|
||||
return [("model.norm.weight", param)]
|
||||
|
||||
try:
|
||||
head_dim = args.kv_channels if args.kv_channels is not None else args.hidden_size // args.num_attention_heads
|
||||
except AttributeError:
|
||||
head_dim = args.hidden_size // args.num_attention_heads
|
||||
value_num_per_group = args.num_attention_heads // args.num_query_groups
|
||||
|
||||
decoder_layers_pattern = r"module\.module\.decoder\.layers\.(\d+)\.(.+)"
|
||||
match = re.match(decoder_layers_pattern, name)
|
||||
if match:
|
||||
layer_idx, rest = match.groups()
|
||||
if rest == "self_attention.linear_proj.weight":
|
||||
return [(f"model.layers.{layer_idx}.self_attn.o_proj.weight", param)]
|
||||
elif rest == "self_attention.linear_qkv.weight":
|
||||
|
||||
param = param.view(args.num_query_groups, -1, head_dim, args.hidden_size)
|
||||
q_param, k_param, v_param = torch.split(param, split_size_or_sections=[value_num_per_group, 1, 1], dim=1)
|
||||
q_param = q_param.reshape(-1, args.hidden_size)
|
||||
k_param = k_param.reshape(-1, args.hidden_size)
|
||||
v_param = v_param.reshape(-1, args.hidden_size)
|
||||
return [
|
||||
(f"model.layers.{layer_idx}.self_attn.q_proj.weight", q_param),
|
||||
(f"model.layers.{layer_idx}.self_attn.k_proj.weight", k_param),
|
||||
(f"model.layers.{layer_idx}.self_attn.v_proj.weight", v_param),
|
||||
]
|
||||
elif rest == "self_attention.linear_qkv.bias":
|
||||
param = param.view(args.num_query_groups, -1)
|
||||
q_bias, k_bias, v_bias = torch.split(
|
||||
param,
|
||||
split_size_or_sections=[value_num_per_group * head_dim, head_dim, head_dim],
|
||||
dim=1,
|
||||
)
|
||||
q_bias = q_bias.contiguous().flatten()
|
||||
k_bias = k_bias.contiguous().flatten()
|
||||
v_bias = v_bias.contiguous().flatten()
|
||||
return [
|
||||
(f"model.layers.{layer_idx}.self_attn.q_proj.bias", q_bias),
|
||||
(f"model.layers.{layer_idx}.self_attn.k_proj.bias", k_bias),
|
||||
(f"model.layers.{layer_idx}.self_attn.v_proj.bias", v_bias),
|
||||
]
|
||||
elif rest == "mlp.linear_fc1.weight":
|
||||
gate_weight, up_weight = param.chunk(2, dim=0)
|
||||
return [
|
||||
(f"model.layers.{layer_idx}.mlp.gate_proj.weight", gate_weight),
|
||||
(f"model.layers.{layer_idx}.mlp.up_proj.weight", up_weight),
|
||||
]
|
||||
elif rest == "mlp.linear_fc2.weight":
|
||||
return [(f"model.layers.{layer_idx}.mlp.down_proj.weight", param)]
|
||||
elif rest == "self_attention.linear_qkv.layer_norm_weight":
|
||||
return [(f"model.layers.{layer_idx}.input_layernorm.weight", param)]
|
||||
elif rest == "mlp.linear_fc1.layer_norm_weight":
|
||||
return [(f"model.layers.{layer_idx}.post_attention_layernorm.weight", param)]
|
||||
|
||||
# qk norm
|
||||
elif rest == "self_attention.q_layernorm.weight":
|
||||
return [(f"model.layers.{layer_idx}.self_attn.q_norm.weight", param)]
|
||||
elif rest == "self_attention.k_layernorm.weight":
|
||||
return [(f"model.layers.{layer_idx}.self_attn.k_norm.weight", param)]
|
||||
|
||||
raise ValueError(f"Unknown parameter name: {name}")
|
||||
145
slime/backends/megatron_utils/megatron_to_hf/qwen3_next.py
Normal file
145
slime/backends/megatron_utils/megatron_to_hf/qwen3_next.py
Normal file
@@ -0,0 +1,145 @@
|
||||
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
import re
|
||||
|
||||
import torch
|
||||
|
||||
|
||||
def convert_qwen3_next_to_hf(args, name, param):
|
||||
if name == "module.module.embedding.word_embeddings.weight":
|
||||
return [("model.embed_tokens.weight", param)]
|
||||
if name == "module.module.output_layer.weight":
|
||||
return [("lm_head.weight", param)]
|
||||
if name == "module.module.decoder.final_layernorm.weight":
|
||||
return [("model.norm.weight", param)]
|
||||
|
||||
try:
|
||||
head_dim = args.kv_channels if args.kv_channels is not None else args.hidden_size // args.num_attention_heads
|
||||
except AttributeError:
|
||||
head_dim = args.hidden_size // args.num_attention_heads
|
||||
value_num_per_group = args.num_attention_heads // args.num_query_groups
|
||||
|
||||
decoder_layers_pattern = r"module\.module\.decoder\.layers\.(\d+)\.(.+)"
|
||||
match = re.match(decoder_layers_pattern, name)
|
||||
if match:
|
||||
layer_idx, rest = match.groups()
|
||||
|
||||
# experts
|
||||
expert_pattern = r"mlp.experts\.(.+)\.weight(\d+)"
|
||||
match = re.match(expert_pattern, rest)
|
||||
if match:
|
||||
rest, expert_idx = match.groups()
|
||||
if rest == "linear_fc1":
|
||||
gate_weight, up_weight = param.chunk(2, dim=0)
|
||||
outputs = [
|
||||
(f"model.layers.{layer_idx}.mlp.experts.{expert_idx}.gate_proj.weight", gate_weight),
|
||||
(f"model.layers.{layer_idx}.mlp.experts.{expert_idx}.up_proj.weight", up_weight),
|
||||
]
|
||||
return outputs
|
||||
elif rest == "linear_fc2":
|
||||
outputs = [
|
||||
(f"model.layers.{layer_idx}.mlp.experts.{expert_idx}.down_proj.weight", param),
|
||||
]
|
||||
return outputs
|
||||
else:
|
||||
raise ValueError(f"Unknown expert parameter name: {name}")
|
||||
|
||||
# shared expert
|
||||
shared_expert_pattern = r"mlp.shared_experts\.(.+)"
|
||||
match = re.match(shared_expert_pattern, rest)
|
||||
if match:
|
||||
rest = match.groups()[0]
|
||||
if rest == "linear_fc1.weight":
|
||||
gate_weight, up_weight = param.chunk(2, dim=0)
|
||||
return [
|
||||
(f"model.layers.{layer_idx}.mlp.shared_expert.gate_proj.weight", gate_weight),
|
||||
(f"model.layers.{layer_idx}.mlp.shared_expert.up_proj.weight", up_weight),
|
||||
]
|
||||
elif rest == "linear_fc2.weight":
|
||||
return [(f"model.layers.{layer_idx}.mlp.shared_expert.down_proj.weight", param)]
|
||||
elif rest == "gate_weight":
|
||||
return [(f"model.layers.{layer_idx}.mlp.shared_expert_gate.weight", param)]
|
||||
else:
|
||||
raise ValueError(f"Unknown shared expert parameter name: {name}")
|
||||
|
||||
if rest == "self_attention.linear_proj.weight":
|
||||
return [(f"model.layers.{layer_idx}.self_attn.o_proj.weight", param)]
|
||||
elif rest == "self_attention.linear_qkv.weight":
|
||||
param = param.view(args.num_query_groups, -1, head_dim, args.hidden_size)
|
||||
q_param, k_param, v_param = torch.split(
|
||||
param, split_size_or_sections=[2 * value_num_per_group, 1, 1], dim=1
|
||||
)
|
||||
q_param = (
|
||||
q_param.reshape(args.num_query_groups, 2, value_num_per_group, head_dim, args.hidden_size)
|
||||
.transpose(1, 2)
|
||||
.reshape(-1, args.hidden_size)
|
||||
)
|
||||
k_param = k_param.reshape(-1, args.hidden_size)
|
||||
v_param = v_param.reshape(-1, args.hidden_size)
|
||||
return [
|
||||
(f"model.layers.{layer_idx}.self_attn.q_proj.weight", q_param),
|
||||
(f"model.layers.{layer_idx}.self_attn.k_proj.weight", k_param),
|
||||
(f"model.layers.{layer_idx}.self_attn.v_proj.weight", v_param),
|
||||
]
|
||||
elif rest == "self_attention.linear_qkv.bias":
|
||||
param = param.view(args.num_query_groups, -1)
|
||||
q_bias, k_bias, v_bias = torch.split(
|
||||
param,
|
||||
split_size_or_sections=[value_num_per_group * head_dim, head_dim, head_dim],
|
||||
dim=1,
|
||||
)
|
||||
q_bias = q_bias.contiguous().flatten()
|
||||
k_bias = k_bias.contiguous().flatten()
|
||||
v_bias = v_bias.contiguous().flatten()
|
||||
return [
|
||||
(f"model.layers.{layer_idx}.self_attn.q_proj.bias", q_bias),
|
||||
(f"model.layers.{layer_idx}.self_attn.k_proj.bias", k_bias),
|
||||
(f"model.layers.{layer_idx}.self_attn.v_proj.bias", v_bias),
|
||||
]
|
||||
elif rest == "mlp.linear_fc1.weight":
|
||||
gate_weight, up_weight = param.chunk(2, dim=0)
|
||||
return [
|
||||
(f"model.layers.{layer_idx}.mlp.gate_proj.weight", gate_weight),
|
||||
(f"model.layers.{layer_idx}.mlp.up_proj.weight", up_weight),
|
||||
]
|
||||
elif rest == "mlp.linear_fc2.weight":
|
||||
return [(f"model.layers.{layer_idx}.mlp.down_proj.weight", param)]
|
||||
elif rest == "self_attention.linear_qkv.layer_norm_weight":
|
||||
return [(f"model.layers.{layer_idx}.input_layernorm.weight", param)]
|
||||
elif rest == "mlp.linear_fc1.layer_norm_weight":
|
||||
return [(f"model.layers.{layer_idx}.post_attention_layernorm.weight", param)]
|
||||
elif rest == "pre_mlp_layernorm.weight":
|
||||
return [(f"model.layers.{layer_idx}.post_attention_layernorm.weight", param)]
|
||||
elif rest == "mlp.router.weight":
|
||||
return [(f"model.layers.{layer_idx}.mlp.gate.weight", param)]
|
||||
elif rest == "mlp.router.expert_bias":
|
||||
return [(f"model.layers.{layer_idx}.mlp.gate.e_score_correction_bias", param)]
|
||||
|
||||
# qk norm
|
||||
elif rest == "self_attention.q_layernorm.weight":
|
||||
return [(f"model.layers.{layer_idx}.self_attn.q_norm.weight", param)]
|
||||
elif rest == "self_attention.k_layernorm.weight":
|
||||
return [(f"model.layers.{layer_idx}.self_attn.k_norm.weight", param)]
|
||||
elif rest.startswith("self_attention.") and rest[len("self_attention.") :] in [
|
||||
"input_layernorm.weight",
|
||||
# linear attn
|
||||
"linear_attn.A_log",
|
||||
"linear_attn.conv1d.weight",
|
||||
"linear_attn.dt_bias",
|
||||
"linear_attn.in_proj_ba.weight",
|
||||
"linear_attn.in_proj_qkvz.weight",
|
||||
"linear_attn.norm.weight",
|
||||
"linear_attn.out_proj.weight",
|
||||
# gated attn
|
||||
"self_attn.k_norm.weight",
|
||||
"self_attn.k_proj.weight",
|
||||
"self_attn.o_proj.weight",
|
||||
"self_attn.q_norm.weight",
|
||||
"self_attn.q_proj.weight",
|
||||
"self_attn.v_proj.weight",
|
||||
]:
|
||||
rest = rest[len("self_attention.") :]
|
||||
return [(f"model.layers.{layer_idx}.{rest}", param)]
|
||||
|
||||
raise ValueError(f"Unknown parameter name: {name}")
|
||||
120
slime/backends/megatron_utils/megatron_to_hf/qwen3moe.py
Normal file
120
slime/backends/megatron_utils/megatron_to_hf/qwen3moe.py
Normal file
@@ -0,0 +1,120 @@
|
||||
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
import re
|
||||
|
||||
import torch
|
||||
|
||||
|
||||
def convert_qwen3moe_to_hf(args, name, param):
|
||||
if name == "module.module.embedding.word_embeddings.weight":
|
||||
return [("model.embed_tokens.weight", param)]
|
||||
if name == "module.module.output_layer.weight":
|
||||
return [("lm_head.weight", param)]
|
||||
if name == "module.module.decoder.final_layernorm.weight":
|
||||
return [("model.norm.weight", param)]
|
||||
|
||||
try:
|
||||
head_dim = args.kv_channels if args.kv_channels is not None else args.hidden_size // args.num_attention_heads
|
||||
except AttributeError:
|
||||
head_dim = args.hidden_size // args.num_attention_heads
|
||||
value_num_per_group = args.num_attention_heads // args.num_query_groups
|
||||
|
||||
decoder_layers_pattern = r"module\.module\.decoder\.layers\.(\d+)\.(.+)"
|
||||
match = re.match(decoder_layers_pattern, name)
|
||||
if match:
|
||||
layer_idx, rest = match.groups()
|
||||
|
||||
# experts
|
||||
expert_pattern = r"mlp.experts\.(.+)\.weight(\d+)"
|
||||
match = re.match(expert_pattern, rest)
|
||||
if match:
|
||||
rest, expert_idx = match.groups()
|
||||
if rest == "linear_fc1":
|
||||
gate_weight, up_weight = param.chunk(2, dim=0)
|
||||
outputs = [
|
||||
(f"model.layers.{layer_idx}.mlp.experts.{expert_idx}.gate_proj.weight", gate_weight),
|
||||
(f"model.layers.{layer_idx}.mlp.experts.{expert_idx}.up_proj.weight", up_weight),
|
||||
]
|
||||
return outputs
|
||||
elif rest == "linear_fc2":
|
||||
outputs = [
|
||||
(f"model.layers.{layer_idx}.mlp.experts.{expert_idx}.down_proj.weight", param),
|
||||
]
|
||||
return outputs
|
||||
else:
|
||||
raise ValueError(f"Unknown expert parameter name: {name}")
|
||||
|
||||
# shared expert
|
||||
shared_expert_pattern = r"mlp.shared_experts\.(.+)"
|
||||
match = re.match(shared_expert_pattern, rest)
|
||||
if match:
|
||||
rest = match.groups()[0]
|
||||
if rest == "linear_fc1.weight":
|
||||
gate_weight, up_weight = param.chunk(2, dim=0)
|
||||
return [
|
||||
(f"model.layers.{layer_idx}.mlp.shared_expert.gate_proj.weight", gate_weight),
|
||||
(f"model.layers.{layer_idx}.mlp.shared_expert.up_proj.weight", up_weight),
|
||||
]
|
||||
elif rest == "linear_fc2.weight":
|
||||
return [(f"model.layers.{layer_idx}.mlp.shared_expert.down_proj.weight", param)]
|
||||
elif rest == "gate_weight":
|
||||
return [(f"model.layers.{layer_idx}.mlp.shared_expert_gate.weight", param)]
|
||||
else:
|
||||
raise ValueError(f"Unknown shared expert parameter name: {name}")
|
||||
|
||||
if rest == "self_attention.linear_proj.weight":
|
||||
return [(f"model.layers.{layer_idx}.self_attn.o_proj.weight", param)]
|
||||
elif rest == "self_attention.linear_qkv.weight":
|
||||
|
||||
param = param.view(args.num_query_groups, -1, head_dim, args.hidden_size)
|
||||
q_param, k_param, v_param = torch.split(param, split_size_or_sections=[value_num_per_group, 1, 1], dim=1)
|
||||
q_param = q_param.reshape(-1, args.hidden_size)
|
||||
k_param = k_param.reshape(-1, args.hidden_size)
|
||||
v_param = v_param.reshape(-1, args.hidden_size)
|
||||
return [
|
||||
(f"model.layers.{layer_idx}.self_attn.q_proj.weight", q_param),
|
||||
(f"model.layers.{layer_idx}.self_attn.k_proj.weight", k_param),
|
||||
(f"model.layers.{layer_idx}.self_attn.v_proj.weight", v_param),
|
||||
]
|
||||
elif rest == "self_attention.linear_qkv.bias":
|
||||
param = param.view(args.num_query_groups, -1)
|
||||
q_bias, k_bias, v_bias = torch.split(
|
||||
param,
|
||||
split_size_or_sections=[value_num_per_group * head_dim, head_dim, head_dim],
|
||||
dim=1,
|
||||
)
|
||||
q_bias = q_bias.contiguous().flatten()
|
||||
k_bias = k_bias.contiguous().flatten()
|
||||
v_bias = v_bias.contiguous().flatten()
|
||||
return [
|
||||
(f"model.layers.{layer_idx}.self_attn.q_proj.bias", q_bias),
|
||||
(f"model.layers.{layer_idx}.self_attn.k_proj.bias", k_bias),
|
||||
(f"model.layers.{layer_idx}.self_attn.v_proj.bias", v_bias),
|
||||
]
|
||||
elif rest == "mlp.linear_fc1.weight":
|
||||
gate_weight, up_weight = param.chunk(2, dim=0)
|
||||
return [
|
||||
(f"model.layers.{layer_idx}.mlp.gate_proj.weight", gate_weight),
|
||||
(f"model.layers.{layer_idx}.mlp.up_proj.weight", up_weight),
|
||||
]
|
||||
elif rest == "mlp.linear_fc2.weight":
|
||||
return [(f"model.layers.{layer_idx}.mlp.down_proj.weight", param)]
|
||||
elif rest == "self_attention.linear_qkv.layer_norm_weight":
|
||||
return [(f"model.layers.{layer_idx}.input_layernorm.weight", param)]
|
||||
elif rest == "mlp.linear_fc1.layer_norm_weight":
|
||||
return [(f"model.layers.{layer_idx}.post_attention_layernorm.weight", param)]
|
||||
elif rest == "pre_mlp_layernorm.weight":
|
||||
return [(f"model.layers.{layer_idx}.post_attention_layernorm.weight", param)]
|
||||
elif rest == "mlp.router.weight":
|
||||
return [(f"model.layers.{layer_idx}.mlp.gate.weight", param)]
|
||||
elif rest == "mlp.router.expert_bias":
|
||||
return [(f"model.layers.{layer_idx}.mlp.gate.e_score_correction_bias", param)]
|
||||
|
||||
# qk norm
|
||||
elif rest == "self_attention.q_layernorm.weight":
|
||||
return [(f"model.layers.{layer_idx}.self_attn.q_norm.weight", param)]
|
||||
elif rest == "self_attention.k_layernorm.weight":
|
||||
return [(f"model.layers.{layer_idx}.self_attn.k_norm.weight", param)]
|
||||
|
||||
raise ValueError(f"Unknown parameter name: {name}")
|
||||
Reference in New Issue
Block a user