初始化项目,由ModelHub XC社区提供模型

Model: ayh015/myLightningOPD
Source: Original Platform
This commit is contained in:
ModelHub XC
2026-08-27 23:50:14 +08:00
commit d4e0a1af66
368 changed files with 559583 additions and 0 deletions

View File

@@ -0,0 +1,88 @@
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: Apache-2.0
from .deepseekv3 import convert_deepseekv3_to_hf
from .glm4 import convert_glm4_to_hf
from .glm4moe import convert_glm4moe_to_hf
from .llama import convert_llama_to_hf
from .mimo import convert_mimo_to_hf
from .processors.padding_remover import remove_padding
from .processors.quantizer import quantize_params
from .qwen2 import convert_qwen2_to_hf
from .qwen3_next import convert_qwen3_next_to_hf
from .qwen3moe import convert_qwen3moe_to_hf
# TODO unify w/ `convert_to_hf`
def postprocess_hf_param(args, megatron_param_name, hf_param_name, param):
param = remove_padding(megatron_param_name, param, args.vocab_size)
# TODO support quant
return param
# TODO optimize code details
def convert_to_hf(args, model_name, name, param, quantization_config=None):
param = remove_padding(name, param, args.vocab_size)
converted_named_tensors = _convert_to_hf_core(args, model_name, name, param)
if not quantization_config:
return converted_named_tensors
return quantize_params(args, name, converted_named_tensors, quantization_config)
# TODO optimize
_cached_tensors = {}
# TODO optimize code details
def _convert_to_hf_core(args, model_name, name, param):
if "glm4moe" in model_name:
converted_named_tensors = convert_glm4moe_to_hf(args, name, param)
elif "glm4" in model_name:
converted_named_tensors = convert_glm4_to_hf(args, name, param)
elif "qwen3moe" in model_name:
converted_named_tensors = convert_qwen3moe_to_hf(args, name, param)
elif "qwen3next" in model_name:
converted_named_tensors = convert_qwen3_next_to_hf(args, name, param)
elif "qwen2" in model_name or "qwen3" in model_name:
converted_named_tensors = convert_qwen2_to_hf(args, name, param)
elif "deepseekv3" in model_name:
converted_named_tensors = convert_deepseekv3_to_hf(args, name, param)
elif "llama" in model_name:
converted_named_tensors = convert_llama_to_hf(args, name, param)
elif "mimo" in model_name:
converted_named_tensors = convert_mimo_to_hf(args, name, param)
else:
raise ValueError(f"Unsupported model: {model_name}")
# to compatible with sglang implementation
if args.q_lora_rank is not None:
old_converted_named_tensors = converted_named_tensors
converted_named_tensors = []
for converted_name, converted_param in old_converted_named_tensors:
if "q_a_proj" in converted_name:
pair_name = converted_name.replace("q_a_proj", "kv_a_proj_with_mqa")
if pair_name in _cached_tensors:
converted_named_tensors += [
(converted_name, converted_param),
(pair_name, _cached_tensors[pair_name]),
]
del _cached_tensors[pair_name]
else:
_cached_tensors[converted_name] = converted_param
elif "kv_a_proj_with_mqa" in converted_name:
pair_name = converted_name.replace("kv_a_proj_with_mqa", "q_a_proj")
if pair_name in _cached_tensors:
converted_named_tensors += [
(converted_name, converted_param),
(pair_name, _cached_tensors[pair_name]),
]
del _cached_tensors[pair_name]
else:
_cached_tensors[converted_name] = converted_param
else:
converted_named_tensors.append((converted_name, converted_param))
return converted_named_tensors

View File

@@ -0,0 +1,132 @@
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: Apache-2.0
import re
import torch
def convert_deepseekv3_to_hf(args, name, param):
if name == "module.module.embedding.word_embeddings.weight":
return [("model.embed_tokens.weight", param)]
if name == "module.module.output_layer.weight":
return [("lm_head.weight", param)]
if name == "module.module.decoder.final_layernorm.weight":
return [("model.norm.weight", param)]
try:
head_dim = args.kv_channels if args.kv_channels is not None else args.hidden_size // args.num_attention_heads
except AttributeError:
head_dim = args.hidden_size // args.num_attention_heads
value_num_per_group = args.num_attention_heads // args.num_query_groups
decoder_layers_pattern = r"module\.module\.decoder\.layers\.(\d+)\.(.+)"
match = re.match(decoder_layers_pattern, name)
if match:
layer_idx, rest = match.groups()
# experts
expert_pattern = r"mlp.experts\.(.+)\.weight(\d+)"
match = re.match(expert_pattern, rest)
if match:
rest, expert_idx = match.groups()
if rest == "linear_fc1":
gate_weight, up_weight = param.chunk(2, dim=0)
outputs = [
(f"model.layers.{layer_idx}.mlp.experts.{expert_idx}.gate_proj.weight", gate_weight),
(f"model.layers.{layer_idx}.mlp.experts.{expert_idx}.up_proj.weight", up_weight),
]
return outputs
elif rest == "linear_fc2":
outputs = [
(f"model.layers.{layer_idx}.mlp.experts.{expert_idx}.down_proj.weight", param),
]
return outputs
else:
raise ValueError(f"Unknown expert parameter name: {name}")
# shared expert
shared_expert_pattern = r"mlp.shared_experts\.(.+)"
match = re.match(shared_expert_pattern, rest)
if match:
rest = match.groups()[0]
if rest == "linear_fc1.weight":
gate_weight, up_weight = param.chunk(2, dim=0)
return [
(f"model.layers.{layer_idx}.mlp.shared_experts.gate_proj.weight", gate_weight),
(f"model.layers.{layer_idx}.mlp.shared_experts.up_proj.weight", up_weight),
]
elif rest == "linear_fc2.weight":
return [(f"model.layers.{layer_idx}.mlp.shared_experts.down_proj.weight", param)]
else:
raise ValueError(f"Unknown shared expert parameter name: {name}")
if rest == "self_attention.linear_proj.weight":
return [(f"model.layers.{layer_idx}.self_attn.o_proj.weight", param)]
elif rest == "self_attention.linear_q_proj.weight":
return [(f"model.layers.{layer_idx}.self_attn.q_proj.weight", param)]
elif rest == "self_attention.linear_q_down_proj.weight":
return [(f"model.layers.{layer_idx}.self_attn.q_a_proj.weight", param)]
elif rest == "self_attention.linear_q_up_proj.layer_norm_weight":
return [(f"model.layers.{layer_idx}.self_attn.q_a_layernorm.weight", param)]
elif rest == "self_attention.linear_q_up_proj.weight":
return [(f"model.layers.{layer_idx}.self_attn.q_b_proj.weight", param)]
elif rest == "self_attention.linear_qkv.bias":
param = param.view(args.num_query_groups, -1)
q_bias, k_bias, v_bias = torch.split(
param,
split_size_or_sections=[value_num_per_group * head_dim, head_dim, head_dim],
dim=1,
)
q_bias = q_bias.contiguous().flatten()
k_bias = k_bias.contiguous().flatten()
v_bias = v_bias.contiguous().flatten()
return [
(f"model.layers.{layer_idx}.self_attn.q_proj.bias", q_bias),
(f"model.layers.{layer_idx}.self_attn.k_proj.bias", k_bias),
(f"model.layers.{layer_idx}.self_attn.v_proj.bias", v_bias),
]
elif rest == "mlp.linear_fc1.weight":
gate_weight, up_weight = param.chunk(2, dim=0)
return [
(f"model.layers.{layer_idx}.mlp.gate_proj.weight", gate_weight),
(f"model.layers.{layer_idx}.mlp.up_proj.weight", up_weight),
]
elif rest == "mlp.linear_fc2.weight":
return [(f"model.layers.{layer_idx}.mlp.down_proj.weight", param)]
elif rest == "self_attention.linear_qkv.layer_norm_weight" or rest == "input_layernorm.weight":
return [(f"model.layers.{layer_idx}.input_layernorm.weight", param)]
elif rest == "mlp.linear_fc1.layer_norm_weight":
return [(f"model.layers.{layer_idx}.post_attention_layernorm.weight", param)]
elif rest == "self_attention.linear_kv_down_proj.weight":
return [(f"model.layers.{layer_idx}.self_attn.kv_a_proj_with_mqa.weight", param)]
elif rest == "self_attention.linear_kv_up_proj.layer_norm_weight":
return [(f"model.layers.{layer_idx}.self_attn.kv_a_layernorm.weight", param)]
elif rest == "self_attention.linear_kv_up_proj.weight":
return [(f"model.layers.{layer_idx}.self_attn.kv_b_proj.weight", param)]
elif rest == "pre_mlp_layernorm.weight":
return [(f"model.layers.{layer_idx}.post_attention_layernorm.weight", param)]
elif rest == "mlp.router.weight":
return [(f"model.layers.{layer_idx}.mlp.gate.weight", param)]
elif rest == "mlp.router.expert_bias":
return [(f"model.layers.{layer_idx}.mlp.gate.e_score_correction_bias", param)]
mtp_layer_pattern = r"module\.module\.mtp\.layers\.(\d+)\.(.+)"
match = re.match(mtp_layer_pattern, name)
if match:
layer_idx, rest = match.groups()
layer_idx = int(layer_idx) + args.num_layers
if rest == "eh_proj.weight":
return [(f"model.layers.{layer_idx}.eh_proj.weight", param)]
elif rest == "enorm.weight":
return [(f"model.layers.{layer_idx}.enorm.weight", param)]
elif rest == "hnorm.weight":
return [(f"model.layers.{layer_idx}.hnorm.weight", param)]
elif rest == "final_layernorm.weight":
return [(f"model.layers.{layer_idx}.shared_head.norm.weight", param)]
else:
name = f"module.module.decoder.layers.{layer_idx}.{rest}"
name = name.replace("transformer_layer.", "")
return convert_deepseekv3_to_hf(args, name, param)
raise ValueError(f"Unknown parameter name: {name}")

View File

@@ -0,0 +1,78 @@
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: Apache-2.0
import re
import torch
def convert_glm4_to_hf(args, name, param):
if name == "module.module.embedding.word_embeddings.weight":
return [("model.embed_tokens.weight", param)]
if name == "module.module.output_layer.weight":
return [("lm_head.weight", param)]
if name == "module.module.decoder.final_layernorm.weight":
return [("model.norm.weight", param)]
try:
head_dim = args.kv_channels if args.kv_channels is not None else args.hidden_size // args.num_attention_heads
except AttributeError:
head_dim = args.hidden_size // args.num_attention_heads
value_num_per_group = args.num_attention_heads // args.num_query_groups
decoder_layers_pattern = r"module\.module\.decoder\.layers\.(\d+)\.(.+)"
match = re.match(decoder_layers_pattern, name)
if match:
layer_idx, rest = match.groups()
if rest == "self_attention.linear_proj.weight":
return [(f"model.layers.{layer_idx}.self_attn.o_proj.weight", param)]
elif rest == "self_attention.linear_qkv.weight":
param = param.view(args.num_query_groups, -1, head_dim, args.hidden_size)
q_param, k_param, v_param = torch.split(param, split_size_or_sections=[value_num_per_group, 1, 1], dim=1)
q_param = q_param.reshape(-1, args.hidden_size)
k_param = k_param.reshape(-1, args.hidden_size)
v_param = v_param.reshape(-1, args.hidden_size)
return [
(f"model.layers.{layer_idx}.self_attn.q_proj.weight", q_param),
(f"model.layers.{layer_idx}.self_attn.k_proj.weight", k_param),
(f"model.layers.{layer_idx}.self_attn.v_proj.weight", v_param),
]
elif rest == "self_attention.linear_qkv.bias":
param = param.view(args.num_query_groups, -1)
q_bias, k_bias, v_bias = torch.split(
param,
split_size_or_sections=[value_num_per_group * head_dim, head_dim, head_dim],
dim=1,
)
q_bias = q_bias.contiguous().flatten()
k_bias = k_bias.contiguous().flatten()
v_bias = v_bias.contiguous().flatten()
return [
(f"model.layers.{layer_idx}.self_attn.q_proj.bias", q_bias),
(f"model.layers.{layer_idx}.self_attn.k_proj.bias", k_bias),
(f"model.layers.{layer_idx}.self_attn.v_proj.bias", v_bias),
]
elif rest == "mlp.linear_fc1.weight":
return [
(f"model.layers.{layer_idx}.mlp.gate_up_proj.weight", param),
]
elif rest == "mlp.linear_fc2.weight":
return [(f"model.layers.{layer_idx}.mlp.down_proj.weight", param)]
elif rest == "self_attention.linear_qkv.layer_norm_weight":
return [(f"model.layers.{layer_idx}.input_layernorm.weight", param)]
elif rest == "mlp.linear_fc1.layer_norm_weight":
return [(f"model.layers.{layer_idx}.post_attention_layernorm.weight", param)]
# qk norm
elif rest == "self_attention.q_layernorm.weight":
return [(f"model.layers.{layer_idx}.self_attn.q_norm.weight", param)]
elif rest == "self_attention.k_layernorm.weight":
return [(f"model.layers.{layer_idx}.self_attn.k_norm.weight", param)]
# sandwitch norm
elif rest == "post_self_attn_layernorm.weight":
return [(f"model.layers.{layer_idx}.post_self_attn_layernorm.weight", param)]
elif rest == "post_mlp_layernorm.weight":
return [(f"model.layers.{layer_idx}.post_mlp_layernorm.weight", param)]
raise ValueError(f"Unknown parameter name: {name}")

View File

@@ -0,0 +1,142 @@
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: Apache-2.0
import re
import torch
def convert_glm4moe_to_hf(args, name, param):
if name == "module.module.embedding.word_embeddings.weight":
return [("model.embed_tokens.weight", param)]
if name == "module.module.output_layer.weight":
return [("lm_head.weight", param)]
if name == "module.module.decoder.final_layernorm.weight":
return [("model.norm.weight", param)]
try:
head_dim = args.kv_channels if args.kv_channels is not None else args.hidden_size // args.num_attention_heads
except AttributeError:
head_dim = args.hidden_size // args.num_attention_heads
value_num_per_group = args.num_attention_heads // args.num_query_groups
decoder_layers_pattern = r"module\.module\.decoder\.layers\.(\d+)\.(.+)"
match = re.match(decoder_layers_pattern, name)
if match:
layer_idx, rest = match.groups()
# experts
expert_pattern = r"mlp.experts\.(.+)\.weight(\d+)"
match = re.match(expert_pattern, rest)
if match:
rest, expert_idx = match.groups()
if rest == "linear_fc1":
gate_weight, up_weight = param.chunk(2, dim=0)
outputs = [
(f"model.layers.{layer_idx}.mlp.experts.{expert_idx}.gate_proj.weight", gate_weight),
(f"model.layers.{layer_idx}.mlp.experts.{expert_idx}.up_proj.weight", up_weight),
]
return outputs
elif rest == "linear_fc2":
outputs = [
(f"model.layers.{layer_idx}.mlp.experts.{expert_idx}.down_proj.weight", param),
]
return outputs
else:
raise ValueError(f"Unknown expert parameter name: {name}")
# shared expert
shared_expert_pattern = r"mlp.shared_experts\.(.+)"
match = re.match(shared_expert_pattern, rest)
if match:
rest = match.groups()[0]
if rest == "linear_fc1.weight":
gate_weight, up_weight = param.chunk(2, dim=0)
return [
(f"model.layers.{layer_idx}.mlp.shared_experts.gate_proj.weight", gate_weight),
(f"model.layers.{layer_idx}.mlp.shared_experts.up_proj.weight", up_weight),
]
elif rest == "linear_fc2.weight":
return [
(f"model.layers.{layer_idx}.mlp.shared_experts.down_proj.weight", param),
]
else:
raise ValueError(f"Unknown shared expert parameter name: {name}")
if rest == "self_attention.linear_proj.weight":
return [(f"model.layers.{layer_idx}.self_attn.o_proj.weight", param)]
elif rest == "self_attention.linear_qkv.weight":
param = param.view(args.num_query_groups, -1, head_dim, args.hidden_size)
q_param, k_param, v_param = torch.split(param, split_size_or_sections=[value_num_per_group, 1, 1], dim=1)
q_param = q_param.reshape(-1, args.hidden_size)
k_param = k_param.reshape(-1, args.hidden_size)
v_param = v_param.reshape(-1, args.hidden_size)
return [
(f"model.layers.{layer_idx}.self_attn.q_proj.weight", q_param),
(f"model.layers.{layer_idx}.self_attn.k_proj.weight", k_param),
(f"model.layers.{layer_idx}.self_attn.v_proj.weight", v_param),
]
elif rest == "self_attention.linear_qkv.bias":
param = param.view(args.num_query_groups, -1)
q_bias, k_bias, v_bias = torch.split(
param,
split_size_or_sections=[value_num_per_group * head_dim, head_dim, head_dim],
dim=1,
)
q_bias = q_bias.contiguous().flatten()
k_bias = k_bias.contiguous().flatten()
v_bias = v_bias.contiguous().flatten()
return [
(f"model.layers.{layer_idx}.self_attn.q_proj.bias", q_bias),
(f"model.layers.{layer_idx}.self_attn.k_proj.bias", k_bias),
(f"model.layers.{layer_idx}.self_attn.v_proj.bias", v_bias),
]
elif rest == "mlp.linear_fc1.weight":
gate_weight, up_weight = param.chunk(2, dim=0)
return [
(f"model.layers.{layer_idx}.mlp.gate_proj.weight", gate_weight),
(f"model.layers.{layer_idx}.mlp.up_proj.weight", up_weight),
]
elif rest == "mlp.linear_fc2.weight":
return [(f"model.layers.{layer_idx}.mlp.down_proj.weight", param)]
elif rest == "self_attention.linear_qkv.layer_norm_weight":
return [(f"model.layers.{layer_idx}.input_layernorm.weight", param)]
elif rest == "mlp.linear_fc1.layer_norm_weight":
return [(f"model.layers.{layer_idx}.post_attention_layernorm.weight", param)]
elif rest == "post_self_attn_layernorm.weight":
return [(f"model.layers.{layer_idx}.post_self_attn_layernorm.weight", param)]
elif rest == "post_mlp_layernorm.weight":
return [(f"model.layers.{layer_idx}.post_mlp_layernorm.weight", param)]
elif rest == "pre_mlp_layernorm.weight":
return [(f"model.layers.{layer_idx}.post_attention_layernorm.weight", param)]
elif rest == "mlp.router.weight":
return [(f"model.layers.{layer_idx}.mlp.gate.weight", param)]
elif rest == "mlp.router.expert_bias":
return [(f"model.layers.{layer_idx}.mlp.gate.e_score_correction_bias", param)]
# qk norm
elif rest == "self_attention.q_layernorm.weight":
return [(f"model.layers.{layer_idx}.self_attn.q_norm.weight", param)]
elif rest == "self_attention.k_layernorm.weight":
return [(f"model.layers.{layer_idx}.self_attn.k_norm.weight", param)]
mtp_layer_pattern = r"module\.module\.mtp\.layers\.(\d+)\.(.+)"
match = re.match(mtp_layer_pattern, name)
if match:
layer_idx, rest = match.groups()
layer_idx = int(layer_idx) + args.num_layers
if rest == "eh_proj.weight":
return [(f"model.layers.{layer_idx}.eh_proj.weight", param)]
elif rest == "enorm.weight":
return [(f"model.layers.{layer_idx}.enorm.weight", param)]
elif rest == "hnorm.weight":
return [(f"model.layers.{layer_idx}.hnorm.weight", param)]
elif rest == "final_layernorm.weight":
return [(f"model.layers.{layer_idx}.shared_head.norm.weight", param)]
else:
name = f"module.module.decoder.layers.{layer_idx}.{rest}"
name = name.replace("transformer_layer.", "")
return convert_glm4moe_to_hf(args, name, param)
raise ValueError(f"Unknown parameter name: {name}")

View File

@@ -0,0 +1,56 @@
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: Apache-2.0
import re
import torch
def convert_llama_to_hf(args, name, param):
if name == "module.module.embedding.word_embeddings.weight":
return [("model.embed_tokens.weight", param)]
if name == "module.module.output_layer.weight":
return [("lm_head.weight", param)]
if name == "module.module.decoder.final_layernorm.weight":
return [("model.norm.weight", param)]
try:
head_dim = args.kv_channels if args.kv_channels is not None else args.hidden_size // args.num_attention_heads
except AttributeError:
head_dim = args.hidden_size // args.num_attention_heads
value_num_per_group = args.num_attention_heads // args.num_query_groups
decoder_layers_pattern = r"module\.module\.decoder\.layers\.(\d+)\.(.+)"
match = re.match(decoder_layers_pattern, name)
if match:
layer_idx, rest = match.groups()
if rest == "self_attention.linear_proj.weight":
return [(f"model.layers.{layer_idx}.self_attn.o_proj.weight", param)]
elif rest == "self_attention.linear_qkv.weight":
# Split QKV weight for Llama
param = param.view(args.num_query_groups, -1, head_dim, args.hidden_size)
q_param, k_param, v_param = torch.split(param, split_size_or_sections=[value_num_per_group, 1, 1], dim=1)
q_param = q_param.reshape(-1, args.hidden_size)
k_param = k_param.reshape(-1, args.hidden_size)
v_param = v_param.reshape(-1, args.hidden_size)
return [
(f"model.layers.{layer_idx}.self_attn.q_proj.weight", q_param),
(f"model.layers.{layer_idx}.self_attn.k_proj.weight", k_param),
(f"model.layers.{layer_idx}.self_attn.v_proj.weight", v_param),
]
elif rest == "mlp.linear_fc1.weight":
# Split gate and up projections for SwiGLU
gate_weight, up_weight = param.chunk(2, dim=0)
return [
(f"model.layers.{layer_idx}.mlp.gate_proj.weight", gate_weight),
(f"model.layers.{layer_idx}.mlp.up_proj.weight", up_weight),
]
elif rest == "mlp.linear_fc2.weight":
return [(f"model.layers.{layer_idx}.mlp.down_proj.weight", param)]
elif rest == "self_attention.linear_qkv.layer_norm_weight":
return [(f"model.layers.{layer_idx}.input_layernorm.weight", param)]
elif rest == "mlp.linear_fc1.layer_norm_weight":
return [(f"model.layers.{layer_idx}.post_attention_layernorm.weight", param)]
elif rest == "pre_mlp_layernorm.weight":
return [(f"model.layers.{layer_idx}.post_attention_layernorm.weight", param)]
raise ValueError(f"Unknown parameter name: {name}")

View File

@@ -0,0 +1,78 @@
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: Apache-2.0
import re
import torch
from .qwen2 import convert_qwen2_to_hf
def convert_mimo_to_hf(args, name, param):
"""
Convert MiMo model parameters from Megatron to HuggingFace format.
MiMo extends Qwen2 with MTP (Multi-Token Prediction) layers.
"""
if "mtp" in name:
return convert_mimo_mtp_param(args, name, param)
return convert_qwen2_to_hf(args, name, param)
def convert_mimo_mtp_param(args, name, param):
"""
Convert MTP layer parameters from Megatron to HuggingFace format.
MTP layers in MiMo contain:
- LayerNorms (token_layernorm, hidden_layernorm, final_layernorm)
- Input projection (input_proj)
- Self attention (reuses Qwen2 attention structure)
- MLP (reuses Qwen2 MLP structure)
Based on MimoBridge._convert_mtp_param logic (reverse mapping)
"""
mtp_pattern = r"module\.module\.mtp\.layers\.(\d+)\.(.+)"
match = re.match(mtp_pattern, name)
if not match:
raise ValueError(f"Invalid MTP parameter name: {name}")
layer_idx, component = match.groups()
# Direct mappings for MTP-specific components (Megatron -> HF)
# Based on MimoBridge direct_name_mapping (reversed)
direct_mappings = {
"enorm.weight": f"model.mtp_layers.{layer_idx}.token_layernorm.weight",
"hnorm.weight": f"model.mtp_layers.{layer_idx}.hidden_layernorm.weight",
"eh_proj.weight": f"model.mtp_layers.{layer_idx}.input_proj.weight",
"final_layernorm.weight": f"model.mtp_layers.{layer_idx}.final_layernorm.weight",
}
if component == "eh_proj.weight":
first_half, second_half = param.chunk(2, dim=1)
param = torch.cat([second_half, first_half], dim=1)
# Check direct mappings first
if component in direct_mappings:
return [(direct_mappings[component], param)]
# Handle transformer_layer components
if component.startswith("transformer_layer."):
# Remove "transformer_layer." prefix
transformer_component = component[len("transformer_layer.") :]
# Create proxy name for reusing existing Qwen2 conversion functions
proxy_name = f"module.module.decoder.layers.{layer_idx}.{transformer_component}"
# Use existing convert_qwen2_to_hf function for transformer components
results = convert_qwen2_to_hf(args, proxy_name, param)
# Replace model.layers with mtp_layers in results
converted_results = []
for hf_name, hf_param in results:
# Replace model.layers.{idx} with mtp_layers.{idx}
hf_name = hf_name.replace(f"model.layers.{layer_idx}", f"model.mtp_layers.{layer_idx}")
converted_results.append((hf_name, hf_param))
return converted_results
raise ValueError(f"Unknown MTP component: {component} in {name}")

View File

@@ -0,0 +1,3 @@
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: Apache-2.0

View File

@@ -0,0 +1,15 @@
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: Apache-2.0
import torch
from slime.backends.megatron_utils.misc_utils import strip_param_name_prefix
def remove_padding(name: str, param: torch.Tensor, vocab_size: int) -> torch.Tensor:
"""
Remove vocab padding: param[:vocab_size] for embedding/output layers, else unchanged.
"""
if strip_param_name_prefix(name) in {"embedding.word_embeddings.weight", "output_layer.weight"}:
return param[:vocab_size]
return param

View File

@@ -0,0 +1,110 @@
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: Apache-2.0
import re
import torch
from slime.utils.fp8_kernel import blockwise_cast_to_fp8_triton
from ...sglang import quant_weight_ue8m0, should_deepgemm_weight_requant_ue8m0, transform_scale_ue8m0
def quantize_params(args, megatron_name, converted_named_params, quantization_config):
if quantization_config is None:
return converted_named_params
assert quantization_config["quant_method"] == "fp8"
assert quantization_config["fmt"] == "e4m3"
assert quantization_config["activation_scheme"] == "dynamic"
weight_block_size = quantization_config.get("weight_block_size", None)
decoder_layers_pattern = r"module\.module\.decoder\.layers\.(\d+)\.(.+)"
match = re.match(decoder_layers_pattern, megatron_name)
if not match:
# check mtp layers
mtp_layer_pattern = r"module\.module\.mtp\.layers\.(\d+)\.(.+)"
match = re.match(mtp_layer_pattern, megatron_name)
if not match:
return converted_named_params
layer_idx, rest = match.groups()
rest = rest.replace("transformer_layer.", "")
else:
layer_idx, rest = match.groups()
# experts
expert_pattern = r"mlp.experts\.(.+)\.weight(\d+)"
match = re.match(expert_pattern, rest)
if match:
rest, expert_idx = match.groups()
if rest in [
"linear_fc1",
"linear_fc2",
]:
quantize_named_params = []
for converted_name, param in converted_named_params:
# skip bf16 weight_scale and input_scale
# TODO: find a clearer way.
if converted_name.endswith("_scale"):
continue
quantize_named_params.extend(_quantize_param(converted_name, param, weight_block_size))
return quantize_named_params
# shared expert
shared_expert_pattern = r"mlp.shared_experts\.(.+)"
match = re.match(shared_expert_pattern, rest)
if match:
rest = match.groups()[0]
if rest in [
"linear_fc1.weight",
"linear_fc2.weight",
]:
quantize_named_params = []
for converted_name, param in converted_named_params:
quantize_named_params.extend(_quantize_param(converted_name, param, weight_block_size))
return quantize_named_params
if rest in [
"self_attention.linear_proj.weight",
"self_attention.linear_qkv.weight",
"mlp.linear_fc1.weight",
"mlp.linear_fc2.weight",
# mla
"self_attention.linear_q_proj.weight",
"self_attention.linear_q_down_proj.weight",
"self_attention.linear_q_up_proj.weight",
"self_attention.linear_kv_down_proj.weight",
"self_attention.linear_kv_up_proj.weight",
]:
quantize_named_params = []
for converted_name, param in converted_named_params:
quantize_named_params.extend(_quantize_param(converted_name, param, weight_block_size))
return quantize_named_params
# for other parameters, we just return the original converted_named_params
return converted_named_params
def _quantize_param(name, weight, weight_block_size):
assert name.endswith(".weight"), f"Expected weight parameter, got {name}"
FP8_MIN = torch.finfo(torch.float8_e4m3fn).min
FP8_MAX = torch.finfo(torch.float8_e4m3fn).max
if weight_block_size is not None:
if should_deepgemm_weight_requant_ue8m0 and should_deepgemm_weight_requant_ue8m0(
weight_block_size=weight_block_size
):
qweight, scale = quant_weight_ue8m0(weight, weight_block_size=weight_block_size)
scale = transform_scale_ue8m0(scale, mn=qweight.shape[-2])
else:
qweight, scale = blockwise_cast_to_fp8_triton(weight, weight_block_size)
scale_name = name.replace(".weight", ".weight_scale_inv")
else:
# per tensor quant
scale = weight.abs().max().clamp(min=1e-12).to(torch.float32) / FP8_MAX
qweight = (weight / scale).clamp(min=FP8_MIN, max=FP8_MAX).to(torch.float8_e4m3fn)
scale = scale.view(1)
scale_name = name.replace(".weight", ".weight_scale")
return [(name, qweight), (scale_name, scale)]

View File

@@ -0,0 +1,74 @@
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: Apache-2.0
import re
import torch
def convert_qwen2_to_hf(args, name, param):
if name == "module.module.embedding.word_embeddings.weight":
return [("model.embed_tokens.weight", param)]
if name == "module.module.output_layer.weight":
return [("lm_head.weight", param)]
if name == "module.module.decoder.final_layernorm.weight":
return [("model.norm.weight", param)]
try:
head_dim = args.kv_channels if args.kv_channels is not None else args.hidden_size // args.num_attention_heads
except AttributeError:
head_dim = args.hidden_size // args.num_attention_heads
value_num_per_group = args.num_attention_heads // args.num_query_groups
decoder_layers_pattern = r"module\.module\.decoder\.layers\.(\d+)\.(.+)"
match = re.match(decoder_layers_pattern, name)
if match:
layer_idx, rest = match.groups()
if rest == "self_attention.linear_proj.weight":
return [(f"model.layers.{layer_idx}.self_attn.o_proj.weight", param)]
elif rest == "self_attention.linear_qkv.weight":
param = param.view(args.num_query_groups, -1, head_dim, args.hidden_size)
q_param, k_param, v_param = torch.split(param, split_size_or_sections=[value_num_per_group, 1, 1], dim=1)
q_param = q_param.reshape(-1, args.hidden_size)
k_param = k_param.reshape(-1, args.hidden_size)
v_param = v_param.reshape(-1, args.hidden_size)
return [
(f"model.layers.{layer_idx}.self_attn.q_proj.weight", q_param),
(f"model.layers.{layer_idx}.self_attn.k_proj.weight", k_param),
(f"model.layers.{layer_idx}.self_attn.v_proj.weight", v_param),
]
elif rest == "self_attention.linear_qkv.bias":
param = param.view(args.num_query_groups, -1)
q_bias, k_bias, v_bias = torch.split(
param,
split_size_or_sections=[value_num_per_group * head_dim, head_dim, head_dim],
dim=1,
)
q_bias = q_bias.contiguous().flatten()
k_bias = k_bias.contiguous().flatten()
v_bias = v_bias.contiguous().flatten()
return [
(f"model.layers.{layer_idx}.self_attn.q_proj.bias", q_bias),
(f"model.layers.{layer_idx}.self_attn.k_proj.bias", k_bias),
(f"model.layers.{layer_idx}.self_attn.v_proj.bias", v_bias),
]
elif rest == "mlp.linear_fc1.weight":
gate_weight, up_weight = param.chunk(2, dim=0)
return [
(f"model.layers.{layer_idx}.mlp.gate_proj.weight", gate_weight),
(f"model.layers.{layer_idx}.mlp.up_proj.weight", up_weight),
]
elif rest == "mlp.linear_fc2.weight":
return [(f"model.layers.{layer_idx}.mlp.down_proj.weight", param)]
elif rest == "self_attention.linear_qkv.layer_norm_weight":
return [(f"model.layers.{layer_idx}.input_layernorm.weight", param)]
elif rest == "mlp.linear_fc1.layer_norm_weight":
return [(f"model.layers.{layer_idx}.post_attention_layernorm.weight", param)]
# qk norm
elif rest == "self_attention.q_layernorm.weight":
return [(f"model.layers.{layer_idx}.self_attn.q_norm.weight", param)]
elif rest == "self_attention.k_layernorm.weight":
return [(f"model.layers.{layer_idx}.self_attn.k_norm.weight", param)]
raise ValueError(f"Unknown parameter name: {name}")

View File

@@ -0,0 +1,145 @@
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: Apache-2.0
import re
import torch
def convert_qwen3_next_to_hf(args, name, param):
if name == "module.module.embedding.word_embeddings.weight":
return [("model.embed_tokens.weight", param)]
if name == "module.module.output_layer.weight":
return [("lm_head.weight", param)]
if name == "module.module.decoder.final_layernorm.weight":
return [("model.norm.weight", param)]
try:
head_dim = args.kv_channels if args.kv_channels is not None else args.hidden_size // args.num_attention_heads
except AttributeError:
head_dim = args.hidden_size // args.num_attention_heads
value_num_per_group = args.num_attention_heads // args.num_query_groups
decoder_layers_pattern = r"module\.module\.decoder\.layers\.(\d+)\.(.+)"
match = re.match(decoder_layers_pattern, name)
if match:
layer_idx, rest = match.groups()
# experts
expert_pattern = r"mlp.experts\.(.+)\.weight(\d+)"
match = re.match(expert_pattern, rest)
if match:
rest, expert_idx = match.groups()
if rest == "linear_fc1":
gate_weight, up_weight = param.chunk(2, dim=0)
outputs = [
(f"model.layers.{layer_idx}.mlp.experts.{expert_idx}.gate_proj.weight", gate_weight),
(f"model.layers.{layer_idx}.mlp.experts.{expert_idx}.up_proj.weight", up_weight),
]
return outputs
elif rest == "linear_fc2":
outputs = [
(f"model.layers.{layer_idx}.mlp.experts.{expert_idx}.down_proj.weight", param),
]
return outputs
else:
raise ValueError(f"Unknown expert parameter name: {name}")
# shared expert
shared_expert_pattern = r"mlp.shared_experts\.(.+)"
match = re.match(shared_expert_pattern, rest)
if match:
rest = match.groups()[0]
if rest == "linear_fc1.weight":
gate_weight, up_weight = param.chunk(2, dim=0)
return [
(f"model.layers.{layer_idx}.mlp.shared_expert.gate_proj.weight", gate_weight),
(f"model.layers.{layer_idx}.mlp.shared_expert.up_proj.weight", up_weight),
]
elif rest == "linear_fc2.weight":
return [(f"model.layers.{layer_idx}.mlp.shared_expert.down_proj.weight", param)]
elif rest == "gate_weight":
return [(f"model.layers.{layer_idx}.mlp.shared_expert_gate.weight", param)]
else:
raise ValueError(f"Unknown shared expert parameter name: {name}")
if rest == "self_attention.linear_proj.weight":
return [(f"model.layers.{layer_idx}.self_attn.o_proj.weight", param)]
elif rest == "self_attention.linear_qkv.weight":
param = param.view(args.num_query_groups, -1, head_dim, args.hidden_size)
q_param, k_param, v_param = torch.split(
param, split_size_or_sections=[2 * value_num_per_group, 1, 1], dim=1
)
q_param = (
q_param.reshape(args.num_query_groups, 2, value_num_per_group, head_dim, args.hidden_size)
.transpose(1, 2)
.reshape(-1, args.hidden_size)
)
k_param = k_param.reshape(-1, args.hidden_size)
v_param = v_param.reshape(-1, args.hidden_size)
return [
(f"model.layers.{layer_idx}.self_attn.q_proj.weight", q_param),
(f"model.layers.{layer_idx}.self_attn.k_proj.weight", k_param),
(f"model.layers.{layer_idx}.self_attn.v_proj.weight", v_param),
]
elif rest == "self_attention.linear_qkv.bias":
param = param.view(args.num_query_groups, -1)
q_bias, k_bias, v_bias = torch.split(
param,
split_size_or_sections=[value_num_per_group * head_dim, head_dim, head_dim],
dim=1,
)
q_bias = q_bias.contiguous().flatten()
k_bias = k_bias.contiguous().flatten()
v_bias = v_bias.contiguous().flatten()
return [
(f"model.layers.{layer_idx}.self_attn.q_proj.bias", q_bias),
(f"model.layers.{layer_idx}.self_attn.k_proj.bias", k_bias),
(f"model.layers.{layer_idx}.self_attn.v_proj.bias", v_bias),
]
elif rest == "mlp.linear_fc1.weight":
gate_weight, up_weight = param.chunk(2, dim=0)
return [
(f"model.layers.{layer_idx}.mlp.gate_proj.weight", gate_weight),
(f"model.layers.{layer_idx}.mlp.up_proj.weight", up_weight),
]
elif rest == "mlp.linear_fc2.weight":
return [(f"model.layers.{layer_idx}.mlp.down_proj.weight", param)]
elif rest == "self_attention.linear_qkv.layer_norm_weight":
return [(f"model.layers.{layer_idx}.input_layernorm.weight", param)]
elif rest == "mlp.linear_fc1.layer_norm_weight":
return [(f"model.layers.{layer_idx}.post_attention_layernorm.weight", param)]
elif rest == "pre_mlp_layernorm.weight":
return [(f"model.layers.{layer_idx}.post_attention_layernorm.weight", param)]
elif rest == "mlp.router.weight":
return [(f"model.layers.{layer_idx}.mlp.gate.weight", param)]
elif rest == "mlp.router.expert_bias":
return [(f"model.layers.{layer_idx}.mlp.gate.e_score_correction_bias", param)]
# qk norm
elif rest == "self_attention.q_layernorm.weight":
return [(f"model.layers.{layer_idx}.self_attn.q_norm.weight", param)]
elif rest == "self_attention.k_layernorm.weight":
return [(f"model.layers.{layer_idx}.self_attn.k_norm.weight", param)]
elif rest.startswith("self_attention.") and rest[len("self_attention.") :] in [
"input_layernorm.weight",
# linear attn
"linear_attn.A_log",
"linear_attn.conv1d.weight",
"linear_attn.dt_bias",
"linear_attn.in_proj_ba.weight",
"linear_attn.in_proj_qkvz.weight",
"linear_attn.norm.weight",
"linear_attn.out_proj.weight",
# gated attn
"self_attn.k_norm.weight",
"self_attn.k_proj.weight",
"self_attn.o_proj.weight",
"self_attn.q_norm.weight",
"self_attn.q_proj.weight",
"self_attn.v_proj.weight",
]:
rest = rest[len("self_attention.") :]
return [(f"model.layers.{layer_idx}.{rest}", param)]
raise ValueError(f"Unknown parameter name: {name}")

View File

@@ -0,0 +1,120 @@
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: Apache-2.0
import re
import torch
def convert_qwen3moe_to_hf(args, name, param):
if name == "module.module.embedding.word_embeddings.weight":
return [("model.embed_tokens.weight", param)]
if name == "module.module.output_layer.weight":
return [("lm_head.weight", param)]
if name == "module.module.decoder.final_layernorm.weight":
return [("model.norm.weight", param)]
try:
head_dim = args.kv_channels if args.kv_channels is not None else args.hidden_size // args.num_attention_heads
except AttributeError:
head_dim = args.hidden_size // args.num_attention_heads
value_num_per_group = args.num_attention_heads // args.num_query_groups
decoder_layers_pattern = r"module\.module\.decoder\.layers\.(\d+)\.(.+)"
match = re.match(decoder_layers_pattern, name)
if match:
layer_idx, rest = match.groups()
# experts
expert_pattern = r"mlp.experts\.(.+)\.weight(\d+)"
match = re.match(expert_pattern, rest)
if match:
rest, expert_idx = match.groups()
if rest == "linear_fc1":
gate_weight, up_weight = param.chunk(2, dim=0)
outputs = [
(f"model.layers.{layer_idx}.mlp.experts.{expert_idx}.gate_proj.weight", gate_weight),
(f"model.layers.{layer_idx}.mlp.experts.{expert_idx}.up_proj.weight", up_weight),
]
return outputs
elif rest == "linear_fc2":
outputs = [
(f"model.layers.{layer_idx}.mlp.experts.{expert_idx}.down_proj.weight", param),
]
return outputs
else:
raise ValueError(f"Unknown expert parameter name: {name}")
# shared expert
shared_expert_pattern = r"mlp.shared_experts\.(.+)"
match = re.match(shared_expert_pattern, rest)
if match:
rest = match.groups()[0]
if rest == "linear_fc1.weight":
gate_weight, up_weight = param.chunk(2, dim=0)
return [
(f"model.layers.{layer_idx}.mlp.shared_expert.gate_proj.weight", gate_weight),
(f"model.layers.{layer_idx}.mlp.shared_expert.up_proj.weight", up_weight),
]
elif rest == "linear_fc2.weight":
return [(f"model.layers.{layer_idx}.mlp.shared_expert.down_proj.weight", param)]
elif rest == "gate_weight":
return [(f"model.layers.{layer_idx}.mlp.shared_expert_gate.weight", param)]
else:
raise ValueError(f"Unknown shared expert parameter name: {name}")
if rest == "self_attention.linear_proj.weight":
return [(f"model.layers.{layer_idx}.self_attn.o_proj.weight", param)]
elif rest == "self_attention.linear_qkv.weight":
param = param.view(args.num_query_groups, -1, head_dim, args.hidden_size)
q_param, k_param, v_param = torch.split(param, split_size_or_sections=[value_num_per_group, 1, 1], dim=1)
q_param = q_param.reshape(-1, args.hidden_size)
k_param = k_param.reshape(-1, args.hidden_size)
v_param = v_param.reshape(-1, args.hidden_size)
return [
(f"model.layers.{layer_idx}.self_attn.q_proj.weight", q_param),
(f"model.layers.{layer_idx}.self_attn.k_proj.weight", k_param),
(f"model.layers.{layer_idx}.self_attn.v_proj.weight", v_param),
]
elif rest == "self_attention.linear_qkv.bias":
param = param.view(args.num_query_groups, -1)
q_bias, k_bias, v_bias = torch.split(
param,
split_size_or_sections=[value_num_per_group * head_dim, head_dim, head_dim],
dim=1,
)
q_bias = q_bias.contiguous().flatten()
k_bias = k_bias.contiguous().flatten()
v_bias = v_bias.contiguous().flatten()
return [
(f"model.layers.{layer_idx}.self_attn.q_proj.bias", q_bias),
(f"model.layers.{layer_idx}.self_attn.k_proj.bias", k_bias),
(f"model.layers.{layer_idx}.self_attn.v_proj.bias", v_bias),
]
elif rest == "mlp.linear_fc1.weight":
gate_weight, up_weight = param.chunk(2, dim=0)
return [
(f"model.layers.{layer_idx}.mlp.gate_proj.weight", gate_weight),
(f"model.layers.{layer_idx}.mlp.up_proj.weight", up_weight),
]
elif rest == "mlp.linear_fc2.weight":
return [(f"model.layers.{layer_idx}.mlp.down_proj.weight", param)]
elif rest == "self_attention.linear_qkv.layer_norm_weight":
return [(f"model.layers.{layer_idx}.input_layernorm.weight", param)]
elif rest == "mlp.linear_fc1.layer_norm_weight":
return [(f"model.layers.{layer_idx}.post_attention_layernorm.weight", param)]
elif rest == "pre_mlp_layernorm.weight":
return [(f"model.layers.{layer_idx}.post_attention_layernorm.weight", param)]
elif rest == "mlp.router.weight":
return [(f"model.layers.{layer_idx}.mlp.gate.weight", param)]
elif rest == "mlp.router.expert_bias":
return [(f"model.layers.{layer_idx}.mlp.gate.e_score_correction_bias", param)]
# qk norm
elif rest == "self_attention.q_layernorm.weight":
return [(f"model.layers.{layer_idx}.self_attn.q_norm.weight", param)]
elif rest == "self_attention.k_layernorm.weight":
return [(f"model.layers.{layer_idx}.self_attn.k_norm.weight", param)]
raise ValueError(f"Unknown parameter name: {name}")