Files
JiRackUltra_1b/jirack_to_gguf_1p5b.py
ModelHub XC 42aa1b5282 初始化项目,由ModelHub XC社区提供模型
Model: CMSManhattan/JiRackUltra_1b
Source: Original Platform
2026-08-25 05:59:19 +08:00

372 lines
15 KiB
Python

# ==============================================================================
# JiRack -> GGUF converter, 1.5B edition (stage 1: .pt -> HuggingFace folder)
# COPYRIGHT (c) 2026 Konstantin Vladimirovich Grabko.
#
# Verified against JiRackTernaryUltra_1b.py [DS1.5-1]:
# vocab_size 151936, hidden 1536, n_layers 28, n_heads 12, n_kv_heads 2,
# head_dim 128 (12*128=1536 -- the converter's hardcoded 128 is correct),
# rope_theta 10000.0 (same as 7B), rms_eps 1e-6,
# tie_word_embeddings = FALSE [DS1.5-2] -- lm_head ships separately, and
# this converter auto-detects that from the presence of lm_head.weight.
#
# Pipeline is two stages:
#
# Stage 1 (THIS SCRIPT, run in venv_ji):
# model.pt -> HF folder (model.safetensors + config.json + tokenizer)
#
# Stage 2 (llama.cpp, run once per model):
# python convert_hf_to_gguf.py <hf_folder> \
# --outfile jirack_1p5b.gguf --outtype bf16
# ./build/bin/llama-quantize jirack_1p5b.gguf \
# jirack_1p5b.Q4_K_M.gguf Q4_K_M
#
# Key points handled here:
# * config.json is derived from the ACTUAL tensor shapes in the checkpoint,
# so vocab (151936 vs 7B's 152064) and any Net2Net-expanded FFN width are
# picked up automatically -- no stock-config copying.
# * lambda_ buffers (ternary fake-quant training machinery) are dropped --
# at inference you run set_lambda(0.0) anyway, so the stored weights ARE
# the full-precision weights; the exported model is a plain Qwen2 dense.
# * Keys: HF naming passes through; JiRack native naming
# (token_emb / blocks.N.* / ffn_w1-w3-w2) is remapped automatically.
#
# EDIT THE THREE PATHS BELOW.
# ==============================================================================
import json
import os
import re
import shutil
import sys
import torch
# ========================= EDIT THESE =========================
CKPT_PATH = "model.pt"
TOKENIZER_DIR = "."
OUTPUT_DIR = "."
# rope_theta cannot be inferred from tensor shapes -- set per base model:
# DeepSeek-R1-Distill-Qwen-1.5B -> 10000.0 (same as 7B)
# DeepSeek-R1-Distill-Qwen-14B -> 1000000.0
# DeepSeek-R1-Distill-Qwen-32B -> 1000000.0
ROPE_THETA = 10000.0
MAX_POSITION = 131072
RMS_NORM_EPS = 1e-6
# Q2_0 = 2-bit ternary {-1, 0, +1} quantization, one fp16 scale per group of
# weights -- the real encoding for BitNet-style ternary weights, once inference
# actually runs true ternary rather than bf16 dense. For now Q4_K_M remains
# the practical choice; the Q2_0 command is just printed ready for later.
EMIT_Q2_0_CMD = True
Q2_0_GROUP = 64 # 64 = mainline llama.cpp, no fork needed.
# ================================================================
# HF Qwen2 key patterns we expect to find (N = layer index)
HF_LAYER_KEYS = [
"model.layers.{n}.self_attn.q_proj.weight",
"model.layers.{n}.self_attn.q_proj.bias",
"model.layers.{n}.self_attn.k_proj.weight",
"model.layers.{n}.self_attn.k_proj.bias",
"model.layers.{n}.self_attn.v_proj.weight",
"model.layers.{n}.self_attn.v_proj.bias",
"model.layers.{n}.self_attn.o_proj.weight",
"model.layers.{n}.mlp.gate_proj.weight",
"model.layers.{n}.mlp.up_proj.weight",
"model.layers.{n}.mlp.down_proj.weight",
"model.layers.{n}.input_layernorm.weight",
"model.layers.{n}.post_attention_layernorm.weight",
]
HF_TOP_KEYS = [
"model.embed_tokens.weight",
"model.norm.weight",
"lm_head.weight",
]
def load_state_dict(path):
print(f"📥 Loading checkpoint: {path}")
ckpt = torch.load(path, map_location="cpu", weights_only=False)
sd = ckpt["model"] if isinstance(ckpt, dict) and "model" in ckpt else ckpt
if not isinstance(sd, dict):
sys.exit("❌ Checkpoint is not a state_dict and has no 'model' key.")
return sd
def drop_training_buffers(sd):
dropped = [k for k in sd if k.endswith("lambda_")]
for k in dropped:
del sd[k]
if dropped:
print(f"🧹 Dropped {len(dropped)} lambda_ buffers (ternary training machinery).")
return sd
def normalize_keys(sd):
"""Pass HF-style keys through; try trivial prefix fixes; else abort with a listing."""
keys = list(sd.keys())
# Case 1: already HF-style
if "model.embed_tokens.weight" in sd:
print("✅ Keys already use HF (Qwen2) naming -- no remap needed.")
return sd
# Case 2: same names but without the leading 'model.' (e.g. 'embed_tokens.weight')
if "embed_tokens.weight" in sd:
print("🔁 Keys look HF-like without the 'model.' prefix -- adding it.")
out = {}
for k, v in sd.items():
if k == "lm_head.weight":
out[k] = v
else:
out["model." + k] = v
if "model.embed_tokens.weight" in out:
return out
# Case 3: JiRack native naming (token_emb / blocks.N.* / ffn_w1-w3-w2)
if "token_emb.weight" in sd and any(k.startswith("blocks.") for k in sd):
print("🔁 JiRack native naming detected -- remapping to HF (Qwen2) keys.")
hidden = sd["token_emb.weight"].shape[1]
block_map = {
"norm1.weight": "input_layernorm.weight",
"norm2.weight": "post_attention_layernorm.weight",
"q_proj.weight": "self_attn.q_proj.weight",
"q_proj.bias": "self_attn.q_proj.bias",
"k_proj.weight": "self_attn.k_proj.weight",
"k_proj.bias": "self_attn.k_proj.bias",
"v_proj.weight": "self_attn.v_proj.weight",
"v_proj.bias": "self_attn.v_proj.bias",
"out_proj.weight": "self_attn.o_proj.weight",
"ffn_w1.weight": "mlp.gate_proj.weight", # SwiGLU gate
"ffn_w3.weight": "mlp.up_proj.weight", # SwiGLU up
"ffn_w2.weight": "mlp.down_proj.weight", # SwiGLU down
}
out = {"model.embed_tokens.weight": sd["token_emb.weight"]}
leftovers = {}
blk_pat = re.compile(r"^blocks\.(\d+)\.(.+)$")
for k, v in sd.items():
if k == "token_emb.weight":
continue
m = blk_pat.match(k)
if m:
idx, sub = m.group(1), m.group(2)
if sub == "out_proj.bias":
sys.exit("❌ out_proj has a bias -- Qwen2 arch has no o_proj "
"bias, this checkpoint isn't Qwen2-compatible as-is.")
if sub not in block_map:
sys.exit(f"❌ Unknown per-block key: {k} -- send this back.")
out[f"model.layers.{idx}.{block_map[sub]}"] = v
else:
leftovers[k] = v
# classify the remaining top-level keys by tensor shape
for k, v in leftovers.items():
shp = tuple(v.shape)
if len(shp) == 1 and shp[0] == hidden:
print(f" final norm : {k} -> model.norm.weight")
out["model.norm.weight"] = v
elif len(shp) == 2 and shp[1] == hidden:
print(f" lm head : {k} -> lm_head.weight")
out["lm_head.weight"] = v
else:
sys.exit(f"❌ Unexplained top-level key: {k} {shp} -- send back.")
if "model.norm.weight" not in out:
sys.exit("❌ No final-norm tensor found (1-D, size=hidden). Send the "
"full key list (the tail beyond the first 80).")
print(f"✅ Remapped {len(out)} tensors to HF naming.")
return out
# Case 4: unknown naming -- print everything and stop
print("❌ Unrecognized key naming scheme. Full key list (first 80):")
for k in keys[:80]:
print(" ", k, tuple(sd[k].shape) if hasattr(sd[k], "shape") else "")
print(f" ... total {len(keys)} keys")
sys.exit(
"\nSend this key list back and I'll add the exact JiRack->HF mapping "
"to normalize_keys()."
)
def infer_config(sd):
"""Derive Qwen2 config.json entirely from tensor shapes."""
embed = sd["model.embed_tokens.weight"]
vocab_size, hidden_size = embed.shape
layer_ids = set()
pat = re.compile(r"^model\.layers\.(\d+)\.")
for k in sd:
m = pat.match(k)
if m:
layer_ids.add(int(m.group(1)))
num_layers = max(layer_ids) + 1
q_w = sd["model.layers.0.self_attn.q_proj.weight"] # [n_heads*head_dim, hidden]
k_w = sd["model.layers.0.self_attn.k_proj.weight"] # [n_kv*head_dim, hidden]
gate = sd["model.layers.0.mlp.gate_proj.weight"] # [intermediate, hidden]
intermediate_size = gate.shape[0]
# Qwen2 1.5B/7B/14B/32B all use head_dim=128 (1.5B: 12*128=1536)
head_dim = 128
num_attention_heads = q_w.shape[0] // head_dim
num_key_value_heads = k_w.shape[0] // head_dim
# sanity: every layer's FFN must have the same (expanded) width
widths = {sd[f"model.layers.{i}.mlp.gate_proj.weight"].shape[0] for i in layer_ids}
if len(widths) != 1:
sys.exit(f"❌ Inconsistent FFN widths across layers: {sorted(widths)}")
tie = "lm_head.weight" not in sd
cfg = {
"architectures": ["Qwen2ForCausalLM"],
"model_type": "qwen2",
"vocab_size": vocab_size,
"hidden_size": hidden_size,
"intermediate_size": intermediate_size,
"num_hidden_layers": num_layers,
"num_attention_heads": num_attention_heads,
"num_key_value_heads": num_key_value_heads,
"hidden_act": "silu",
"max_position_embeddings": MAX_POSITION,
"rms_norm_eps": RMS_NORM_EPS,
"rope_theta": ROPE_THETA,
"tie_word_embeddings": tie,
"torch_dtype": "bfloat16",
"use_cache": True,
"bos_token_id": 151646,
"eos_token_id": 151643,
}
print("🧾 Inferred config from tensor shapes:")
for k in ("vocab_size", "hidden_size", "intermediate_size", "num_hidden_layers",
"num_attention_heads", "num_key_value_heads", "tie_word_embeddings"):
print(f" {k} = {cfg[k]}")
print(f" rope_theta = {ROPE_THETA} (from the EDIT block -- verify for this base model!)")
return cfg
def save_hf(sd, cfg):
os.makedirs(OUTPUT_DIR, exist_ok=True)
device = "cuda" if torch.cuda.is_available() else "cpu"
print(f"🔄 Casting weights to bf16 on {device.upper()} ...")
for k in sd:
t = sd[k]
if torch.is_tensor(t) and t.is_floating_point():
sd[k] = t.to(device=device, dtype=torch.bfloat16).cpu().contiguous()
try:
from safetensors.torch import save_file
# single-file safetensors; llama.cpp's converter handles it fine
path = os.path.join(OUTPUT_DIR, "model.safetensors")
print(f"💾 Saving {path} ...")
save_file(sd, path, metadata={"format": "pt"})
except ImportError:
# fallback: pytorch_model.bin, also accepted by convert_hf_to_gguf.py
path = os.path.join(OUTPUT_DIR, "pytorch_model.bin")
print(f"⚠️ safetensors not installed -- saving {path} instead (also works).")
torch.save(sd, path)
with open(os.path.join(OUTPUT_DIR, "config.json"), "w") as f:
json.dump(cfg, f, indent=2)
with open(os.path.join(OUTPUT_DIR, "generation_config.json"), "w") as f:
json.dump({"bos_token_id": cfg["bos_token_id"],
"eos_token_id": cfg["eos_token_id"],
"do_sample": True, "temperature": 0.6, "top_p": 0.95}, f, indent=2)
print(f"📎 Copying tokenizer from {TOKENIZER_DIR} ...")
same_dir = os.path.abspath(TOKENIZER_DIR) == os.path.abspath(OUTPUT_DIR)
if same_dir:
print(" TOKENIZER_DIR == OUTPUT_DIR -- tokenizer files are already in "
"place, skipping copy.")
copied = sum(
1 for name in os.listdir(TOKENIZER_DIR)
if name.startswith(("tokenizer", "special_tokens", "added_tokens",
"vocab", "merges", "chat_template"))
)
else:
copied = 0
for name in os.listdir(TOKENIZER_DIR):
if name.startswith(("tokenizer", "special_tokens", "added_tokens", "vocab", "merges", "chat_template")):
shutil.copy2(os.path.join(TOKENIZER_DIR, name), os.path.join(OUTPUT_DIR, name))
copied += 1
if copied == 0:
sys.exit(f"❌ No tokenizer files found in {TOKENIZER_DIR}")
print(f" copied {copied} tokenizer files.")
def verify(cfg):
"""Cross-check tokenizer length vs embedding rows."""
try:
from transformers import AutoTokenizer
tok = AutoTokenizer.from_pretrained(OUTPUT_DIR)
n = len(tok)
rows = cfg["vocab_size"]
if n > rows:
sys.exit(f"❌ Tokenizer has {n} tokens but embedding matrix only {rows} rows -- "
f"resize the checkpoint before converting.")
print(f"✅ Tokenizer check: {n} tokens <= {rows} embedding rows "
f"({rows - n} spare rows).")
except Exception as e:
print(f"⚠️ Could not verify tokenizer ({e}) -- continuing anyway.")
def main():
if not os.path.exists(CKPT_PATH):
sys.exit(f"❌ {CKPT_PATH} not found")
sd = load_state_dict(CKPT_PATH)
sd = drop_training_buffers(sd)
sd = normalize_keys(sd)
cfg = infer_config(sd)
save_hf(sd, cfg)
verify(cfg)
print("\n" + "=" * 78)
print("✅ Stage 1 done. HF model at:", OUTPUT_DIR)
print("=" * 78)
out_norm = OUTPUT_DIR.rstrip("/")
gguf_base = "jirack_1p5b" if out_norm in ("", ".") else out_norm
q2_0_block = ""
if EMIT_Q2_0_CMD:
suffix = "Q2_0" if Q2_0_GROUP == 64 else f"Q2_0_g{Q2_0_GROUP}"
fork_note = (
"group-64 is in mainline llama.cpp -- no fork needed, CPU/Metal ready."
if Q2_0_GROUP == 64 else
"group-128 needs a CUDA fork -- not needed on CPU-only."
)
q2_0_block = """
Ternary quantization (Q2_0, 2 bits/weight, {{-1,0,+1}} + fp16 group scale --
this is the real encoding for BitNet-style ternary weights, once your model
actually runs true ternary at inference rather than bf16 dense):
{fork_note}
./build/bin/llama-quantize {gguf} {gguf_q2} {suffix}
""".format(
fork_note=fork_note,
gguf=gguf_base + ".gguf",
gguf_q2=gguf_base + f".{suffix}.gguf",
suffix=suffix,
)
print("""
Stage 2 -- make the GGUF (one-time llama.cpp setup, then per model):
git clone https://github.com/ggml-org/llama.cpp /mnt/nfs_share/llama.cpp
cd /mnt/nfs_share/llama.cpp
pip install -r requirements.txt
python convert_hf_to_gguf.py {out} \\
--outfile {gguf} --outtype bf16
Optional dense quantization (build llama.cpp first: cmake -B build && cmake --build build -j):
./build/bin/llama-quantize {gguf} {gguf_q} Q4_K_M
{q2_0_block}""".format(
out=OUTPUT_DIR,
gguf=gguf_base + ".gguf",
gguf_q=gguf_base + ".Q4_K_M.gguf",
q2_0_block=q2_0_block,
))
if __name__ == "__main__":
main()