初始化项目,由ModelHub XC社区提供模型
Model: CMSManhattan/JiRackUltra_1b Source: Original Platform
This commit is contained in:
371
jirack_to_gguf_1p5b.py
Normal file
371
jirack_to_gguf_1p5b.py
Normal file
@@ -0,0 +1,371 @@
|
||||
# ==============================================================================
|
||||
# JiRack -> GGUF converter, 1.5B edition (stage 1: .pt -> HuggingFace folder)
|
||||
# COPYRIGHT (c) 2026 Konstantin Vladimirovich Grabko.
|
||||
#
|
||||
# Verified against JiRackTernaryUltra_1b.py [DS1.5-1]:
|
||||
# vocab_size 151936, hidden 1536, n_layers 28, n_heads 12, n_kv_heads 2,
|
||||
# head_dim 128 (12*128=1536 -- the converter's hardcoded 128 is correct),
|
||||
# rope_theta 10000.0 (same as 7B), rms_eps 1e-6,
|
||||
# tie_word_embeddings = FALSE [DS1.5-2] -- lm_head ships separately, and
|
||||
# this converter auto-detects that from the presence of lm_head.weight.
|
||||
#
|
||||
# Pipeline is two stages:
|
||||
#
|
||||
# Stage 1 (THIS SCRIPT, run in venv_ji):
|
||||
# model.pt -> HF folder (model.safetensors + config.json + tokenizer)
|
||||
#
|
||||
# Stage 2 (llama.cpp, run once per model):
|
||||
# python convert_hf_to_gguf.py <hf_folder> \
|
||||
# --outfile jirack_1p5b.gguf --outtype bf16
|
||||
# ./build/bin/llama-quantize jirack_1p5b.gguf \
|
||||
# jirack_1p5b.Q4_K_M.gguf Q4_K_M
|
||||
#
|
||||
# Key points handled here:
|
||||
# * config.json is derived from the ACTUAL tensor shapes in the checkpoint,
|
||||
# so vocab (151936 vs 7B's 152064) and any Net2Net-expanded FFN width are
|
||||
# picked up automatically -- no stock-config copying.
|
||||
# * lambda_ buffers (ternary fake-quant training machinery) are dropped --
|
||||
# at inference you run set_lambda(0.0) anyway, so the stored weights ARE
|
||||
# the full-precision weights; the exported model is a plain Qwen2 dense.
|
||||
# * Keys: HF naming passes through; JiRack native naming
|
||||
# (token_emb / blocks.N.* / ffn_w1-w3-w2) is remapped automatically.
|
||||
#
|
||||
# EDIT THE THREE PATHS BELOW.
|
||||
# ==============================================================================
|
||||
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
import sys
|
||||
|
||||
import torch
|
||||
|
||||
# ========================= EDIT THESE =========================
|
||||
CKPT_PATH = "model.pt"
|
||||
TOKENIZER_DIR = "."
|
||||
OUTPUT_DIR = "."
|
||||
# rope_theta cannot be inferred from tensor shapes -- set per base model:
|
||||
# DeepSeek-R1-Distill-Qwen-1.5B -> 10000.0 (same as 7B)
|
||||
# DeepSeek-R1-Distill-Qwen-14B -> 1000000.0
|
||||
# DeepSeek-R1-Distill-Qwen-32B -> 1000000.0
|
||||
ROPE_THETA = 10000.0
|
||||
MAX_POSITION = 131072
|
||||
RMS_NORM_EPS = 1e-6
|
||||
|
||||
# Q2_0 = 2-bit ternary {-1, 0, +1} quantization, one fp16 scale per group of
|
||||
# weights -- the real encoding for BitNet-style ternary weights, once inference
|
||||
# actually runs true ternary rather than bf16 dense. For now Q4_K_M remains
|
||||
# the practical choice; the Q2_0 command is just printed ready for later.
|
||||
EMIT_Q2_0_CMD = True
|
||||
Q2_0_GROUP = 64 # 64 = mainline llama.cpp, no fork needed.
|
||||
# ================================================================
|
||||
|
||||
# HF Qwen2 key patterns we expect to find (N = layer index)
|
||||
HF_LAYER_KEYS = [
|
||||
"model.layers.{n}.self_attn.q_proj.weight",
|
||||
"model.layers.{n}.self_attn.q_proj.bias",
|
||||
"model.layers.{n}.self_attn.k_proj.weight",
|
||||
"model.layers.{n}.self_attn.k_proj.bias",
|
||||
"model.layers.{n}.self_attn.v_proj.weight",
|
||||
"model.layers.{n}.self_attn.v_proj.bias",
|
||||
"model.layers.{n}.self_attn.o_proj.weight",
|
||||
"model.layers.{n}.mlp.gate_proj.weight",
|
||||
"model.layers.{n}.mlp.up_proj.weight",
|
||||
"model.layers.{n}.mlp.down_proj.weight",
|
||||
"model.layers.{n}.input_layernorm.weight",
|
||||
"model.layers.{n}.post_attention_layernorm.weight",
|
||||
]
|
||||
HF_TOP_KEYS = [
|
||||
"model.embed_tokens.weight",
|
||||
"model.norm.weight",
|
||||
"lm_head.weight",
|
||||
]
|
||||
|
||||
|
||||
def load_state_dict(path):
|
||||
print(f"📥 Loading checkpoint: {path}")
|
||||
ckpt = torch.load(path, map_location="cpu", weights_only=False)
|
||||
sd = ckpt["model"] if isinstance(ckpt, dict) and "model" in ckpt else ckpt
|
||||
if not isinstance(sd, dict):
|
||||
sys.exit("❌ Checkpoint is not a state_dict and has no 'model' key.")
|
||||
return sd
|
||||
|
||||
|
||||
def drop_training_buffers(sd):
|
||||
dropped = [k for k in sd if k.endswith("lambda_")]
|
||||
for k in dropped:
|
||||
del sd[k]
|
||||
if dropped:
|
||||
print(f"🧹 Dropped {len(dropped)} lambda_ buffers (ternary training machinery).")
|
||||
return sd
|
||||
|
||||
|
||||
def normalize_keys(sd):
|
||||
"""Pass HF-style keys through; try trivial prefix fixes; else abort with a listing."""
|
||||
keys = list(sd.keys())
|
||||
|
||||
# Case 1: already HF-style
|
||||
if "model.embed_tokens.weight" in sd:
|
||||
print("✅ Keys already use HF (Qwen2) naming -- no remap needed.")
|
||||
return sd
|
||||
|
||||
# Case 2: same names but without the leading 'model.' (e.g. 'embed_tokens.weight')
|
||||
if "embed_tokens.weight" in sd:
|
||||
print("🔁 Keys look HF-like without the 'model.' prefix -- adding it.")
|
||||
out = {}
|
||||
for k, v in sd.items():
|
||||
if k == "lm_head.weight":
|
||||
out[k] = v
|
||||
else:
|
||||
out["model." + k] = v
|
||||
if "model.embed_tokens.weight" in out:
|
||||
return out
|
||||
|
||||
# Case 3: JiRack native naming (token_emb / blocks.N.* / ffn_w1-w3-w2)
|
||||
if "token_emb.weight" in sd and any(k.startswith("blocks.") for k in sd):
|
||||
print("🔁 JiRack native naming detected -- remapping to HF (Qwen2) keys.")
|
||||
hidden = sd["token_emb.weight"].shape[1]
|
||||
block_map = {
|
||||
"norm1.weight": "input_layernorm.weight",
|
||||
"norm2.weight": "post_attention_layernorm.weight",
|
||||
"q_proj.weight": "self_attn.q_proj.weight",
|
||||
"q_proj.bias": "self_attn.q_proj.bias",
|
||||
"k_proj.weight": "self_attn.k_proj.weight",
|
||||
"k_proj.bias": "self_attn.k_proj.bias",
|
||||
"v_proj.weight": "self_attn.v_proj.weight",
|
||||
"v_proj.bias": "self_attn.v_proj.bias",
|
||||
"out_proj.weight": "self_attn.o_proj.weight",
|
||||
"ffn_w1.weight": "mlp.gate_proj.weight", # SwiGLU gate
|
||||
"ffn_w3.weight": "mlp.up_proj.weight", # SwiGLU up
|
||||
"ffn_w2.weight": "mlp.down_proj.weight", # SwiGLU down
|
||||
}
|
||||
out = {"model.embed_tokens.weight": sd["token_emb.weight"]}
|
||||
leftovers = {}
|
||||
blk_pat = re.compile(r"^blocks\.(\d+)\.(.+)$")
|
||||
for k, v in sd.items():
|
||||
if k == "token_emb.weight":
|
||||
continue
|
||||
m = blk_pat.match(k)
|
||||
if m:
|
||||
idx, sub = m.group(1), m.group(2)
|
||||
if sub == "out_proj.bias":
|
||||
sys.exit("❌ out_proj has a bias -- Qwen2 arch has no o_proj "
|
||||
"bias, this checkpoint isn't Qwen2-compatible as-is.")
|
||||
if sub not in block_map:
|
||||
sys.exit(f"❌ Unknown per-block key: {k} -- send this back.")
|
||||
out[f"model.layers.{idx}.{block_map[sub]}"] = v
|
||||
else:
|
||||
leftovers[k] = v
|
||||
# classify the remaining top-level keys by tensor shape
|
||||
for k, v in leftovers.items():
|
||||
shp = tuple(v.shape)
|
||||
if len(shp) == 1 and shp[0] == hidden:
|
||||
print(f" final norm : {k} -> model.norm.weight")
|
||||
out["model.norm.weight"] = v
|
||||
elif len(shp) == 2 and shp[1] == hidden:
|
||||
print(f" lm head : {k} -> lm_head.weight")
|
||||
out["lm_head.weight"] = v
|
||||
else:
|
||||
sys.exit(f"❌ Unexplained top-level key: {k} {shp} -- send back.")
|
||||
if "model.norm.weight" not in out:
|
||||
sys.exit("❌ No final-norm tensor found (1-D, size=hidden). Send the "
|
||||
"full key list (the tail beyond the first 80).")
|
||||
print(f"✅ Remapped {len(out)} tensors to HF naming.")
|
||||
return out
|
||||
|
||||
# Case 4: unknown naming -- print everything and stop
|
||||
print("❌ Unrecognized key naming scheme. Full key list (first 80):")
|
||||
for k in keys[:80]:
|
||||
print(" ", k, tuple(sd[k].shape) if hasattr(sd[k], "shape") else "")
|
||||
print(f" ... total {len(keys)} keys")
|
||||
sys.exit(
|
||||
"\nSend this key list back and I'll add the exact JiRack->HF mapping "
|
||||
"to normalize_keys()."
|
||||
)
|
||||
|
||||
|
||||
def infer_config(sd):
|
||||
"""Derive Qwen2 config.json entirely from tensor shapes."""
|
||||
embed = sd["model.embed_tokens.weight"]
|
||||
vocab_size, hidden_size = embed.shape
|
||||
|
||||
layer_ids = set()
|
||||
pat = re.compile(r"^model\.layers\.(\d+)\.")
|
||||
for k in sd:
|
||||
m = pat.match(k)
|
||||
if m:
|
||||
layer_ids.add(int(m.group(1)))
|
||||
num_layers = max(layer_ids) + 1
|
||||
|
||||
q_w = sd["model.layers.0.self_attn.q_proj.weight"] # [n_heads*head_dim, hidden]
|
||||
k_w = sd["model.layers.0.self_attn.k_proj.weight"] # [n_kv*head_dim, hidden]
|
||||
gate = sd["model.layers.0.mlp.gate_proj.weight"] # [intermediate, hidden]
|
||||
intermediate_size = gate.shape[0]
|
||||
|
||||
# Qwen2 1.5B/7B/14B/32B all use head_dim=128 (1.5B: 12*128=1536)
|
||||
head_dim = 128
|
||||
num_attention_heads = q_w.shape[0] // head_dim
|
||||
num_key_value_heads = k_w.shape[0] // head_dim
|
||||
|
||||
# sanity: every layer's FFN must have the same (expanded) width
|
||||
widths = {sd[f"model.layers.{i}.mlp.gate_proj.weight"].shape[0] for i in layer_ids}
|
||||
if len(widths) != 1:
|
||||
sys.exit(f"❌ Inconsistent FFN widths across layers: {sorted(widths)}")
|
||||
|
||||
tie = "lm_head.weight" not in sd
|
||||
cfg = {
|
||||
"architectures": ["Qwen2ForCausalLM"],
|
||||
"model_type": "qwen2",
|
||||
"vocab_size": vocab_size,
|
||||
"hidden_size": hidden_size,
|
||||
"intermediate_size": intermediate_size,
|
||||
"num_hidden_layers": num_layers,
|
||||
"num_attention_heads": num_attention_heads,
|
||||
"num_key_value_heads": num_key_value_heads,
|
||||
"hidden_act": "silu",
|
||||
"max_position_embeddings": MAX_POSITION,
|
||||
"rms_norm_eps": RMS_NORM_EPS,
|
||||
"rope_theta": ROPE_THETA,
|
||||
"tie_word_embeddings": tie,
|
||||
"torch_dtype": "bfloat16",
|
||||
"use_cache": True,
|
||||
"bos_token_id": 151646,
|
||||
"eos_token_id": 151643,
|
||||
}
|
||||
print("🧾 Inferred config from tensor shapes:")
|
||||
for k in ("vocab_size", "hidden_size", "intermediate_size", "num_hidden_layers",
|
||||
"num_attention_heads", "num_key_value_heads", "tie_word_embeddings"):
|
||||
print(f" {k} = {cfg[k]}")
|
||||
print(f" rope_theta = {ROPE_THETA} (from the EDIT block -- verify for this base model!)")
|
||||
return cfg
|
||||
|
||||
|
||||
def save_hf(sd, cfg):
|
||||
os.makedirs(OUTPUT_DIR, exist_ok=True)
|
||||
|
||||
device = "cuda" if torch.cuda.is_available() else "cpu"
|
||||
print(f"🔄 Casting weights to bf16 on {device.upper()} ...")
|
||||
for k in sd:
|
||||
t = sd[k]
|
||||
if torch.is_tensor(t) and t.is_floating_point():
|
||||
sd[k] = t.to(device=device, dtype=torch.bfloat16).cpu().contiguous()
|
||||
|
||||
try:
|
||||
from safetensors.torch import save_file
|
||||
# single-file safetensors; llama.cpp's converter handles it fine
|
||||
path = os.path.join(OUTPUT_DIR, "model.safetensors")
|
||||
print(f"💾 Saving {path} ...")
|
||||
save_file(sd, path, metadata={"format": "pt"})
|
||||
except ImportError:
|
||||
# fallback: pytorch_model.bin, also accepted by convert_hf_to_gguf.py
|
||||
path = os.path.join(OUTPUT_DIR, "pytorch_model.bin")
|
||||
print(f"⚠️ safetensors not installed -- saving {path} instead (also works).")
|
||||
torch.save(sd, path)
|
||||
|
||||
with open(os.path.join(OUTPUT_DIR, "config.json"), "w") as f:
|
||||
json.dump(cfg, f, indent=2)
|
||||
with open(os.path.join(OUTPUT_DIR, "generation_config.json"), "w") as f:
|
||||
json.dump({"bos_token_id": cfg["bos_token_id"],
|
||||
"eos_token_id": cfg["eos_token_id"],
|
||||
"do_sample": True, "temperature": 0.6, "top_p": 0.95}, f, indent=2)
|
||||
|
||||
print(f"📎 Copying tokenizer from {TOKENIZER_DIR} ...")
|
||||
same_dir = os.path.abspath(TOKENIZER_DIR) == os.path.abspath(OUTPUT_DIR)
|
||||
if same_dir:
|
||||
print(" TOKENIZER_DIR == OUTPUT_DIR -- tokenizer files are already in "
|
||||
"place, skipping copy.")
|
||||
copied = sum(
|
||||
1 for name in os.listdir(TOKENIZER_DIR)
|
||||
if name.startswith(("tokenizer", "special_tokens", "added_tokens",
|
||||
"vocab", "merges", "chat_template"))
|
||||
)
|
||||
else:
|
||||
copied = 0
|
||||
for name in os.listdir(TOKENIZER_DIR):
|
||||
if name.startswith(("tokenizer", "special_tokens", "added_tokens", "vocab", "merges", "chat_template")):
|
||||
shutil.copy2(os.path.join(TOKENIZER_DIR, name), os.path.join(OUTPUT_DIR, name))
|
||||
copied += 1
|
||||
if copied == 0:
|
||||
sys.exit(f"❌ No tokenizer files found in {TOKENIZER_DIR}")
|
||||
print(f" copied {copied} tokenizer files.")
|
||||
|
||||
|
||||
def verify(cfg):
|
||||
"""Cross-check tokenizer length vs embedding rows."""
|
||||
try:
|
||||
from transformers import AutoTokenizer
|
||||
tok = AutoTokenizer.from_pretrained(OUTPUT_DIR)
|
||||
n = len(tok)
|
||||
rows = cfg["vocab_size"]
|
||||
if n > rows:
|
||||
sys.exit(f"❌ Tokenizer has {n} tokens but embedding matrix only {rows} rows -- "
|
||||
f"resize the checkpoint before converting.")
|
||||
print(f"✅ Tokenizer check: {n} tokens <= {rows} embedding rows "
|
||||
f"({rows - n} spare rows).")
|
||||
except Exception as e:
|
||||
print(f"⚠️ Could not verify tokenizer ({e}) -- continuing anyway.")
|
||||
|
||||
|
||||
def main():
|
||||
if not os.path.exists(CKPT_PATH):
|
||||
sys.exit(f"❌ {CKPT_PATH} not found")
|
||||
sd = load_state_dict(CKPT_PATH)
|
||||
sd = drop_training_buffers(sd)
|
||||
sd = normalize_keys(sd)
|
||||
cfg = infer_config(sd)
|
||||
save_hf(sd, cfg)
|
||||
verify(cfg)
|
||||
|
||||
print("\n" + "=" * 78)
|
||||
print("✅ Stage 1 done. HF model at:", OUTPUT_DIR)
|
||||
print("=" * 78)
|
||||
|
||||
out_norm = OUTPUT_DIR.rstrip("/")
|
||||
gguf_base = "jirack_1p5b" if out_norm in ("", ".") else out_norm
|
||||
|
||||
q2_0_block = ""
|
||||
if EMIT_Q2_0_CMD:
|
||||
suffix = "Q2_0" if Q2_0_GROUP == 64 else f"Q2_0_g{Q2_0_GROUP}"
|
||||
fork_note = (
|
||||
"group-64 is in mainline llama.cpp -- no fork needed, CPU/Metal ready."
|
||||
if Q2_0_GROUP == 64 else
|
||||
"group-128 needs a CUDA fork -- not needed on CPU-only."
|
||||
)
|
||||
q2_0_block = """
|
||||
Ternary quantization (Q2_0, 2 bits/weight, {{-1,0,+1}} + fp16 group scale --
|
||||
this is the real encoding for BitNet-style ternary weights, once your model
|
||||
actually runs true ternary at inference rather than bf16 dense):
|
||||
{fork_note}
|
||||
|
||||
./build/bin/llama-quantize {gguf} {gguf_q2} {suffix}
|
||||
""".format(
|
||||
fork_note=fork_note,
|
||||
gguf=gguf_base + ".gguf",
|
||||
gguf_q2=gguf_base + f".{suffix}.gguf",
|
||||
suffix=suffix,
|
||||
)
|
||||
|
||||
print("""
|
||||
Stage 2 -- make the GGUF (one-time llama.cpp setup, then per model):
|
||||
|
||||
git clone https://github.com/ggml-org/llama.cpp /mnt/nfs_share/llama.cpp
|
||||
cd /mnt/nfs_share/llama.cpp
|
||||
pip install -r requirements.txt
|
||||
|
||||
python convert_hf_to_gguf.py {out} \\
|
||||
--outfile {gguf} --outtype bf16
|
||||
|
||||
Optional dense quantization (build llama.cpp first: cmake -B build && cmake --build build -j):
|
||||
|
||||
./build/bin/llama-quantize {gguf} {gguf_q} Q4_K_M
|
||||
{q2_0_block}""".format(
|
||||
out=OUTPUT_DIR,
|
||||
gguf=gguf_base + ".gguf",
|
||||
gguf_q=gguf_base + ".Q4_K_M.gguf",
|
||||
q2_0_block=q2_0_block,
|
||||
))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user