# ============================================================================== # JiRack -> GGUF converter, 1.5B edition (stage 1: .pt -> HuggingFace folder) # COPYRIGHT (c) 2026 Konstantin Vladimirovich Grabko. # # Verified against JiRackTernaryUltra_1b.py [DS1.5-1]: # vocab_size 151936, hidden 1536, n_layers 28, n_heads 12, n_kv_heads 2, # head_dim 128 (12*128=1536 -- the converter's hardcoded 128 is correct), # rope_theta 10000.0 (same as 7B), rms_eps 1e-6, # tie_word_embeddings = FALSE [DS1.5-2] -- lm_head ships separately, and # this converter auto-detects that from the presence of lm_head.weight. # # Pipeline is two stages: # # Stage 1 (THIS SCRIPT, run in venv_ji): # model.pt -> HF folder (model.safetensors + config.json + tokenizer) # # Stage 2 (llama.cpp, run once per model): # python convert_hf_to_gguf.py \ # --outfile jirack_1p5b.gguf --outtype bf16 # ./build/bin/llama-quantize jirack_1p5b.gguf \ # jirack_1p5b.Q4_K_M.gguf Q4_K_M # # Key points handled here: # * config.json is derived from the ACTUAL tensor shapes in the checkpoint, # so vocab (151936 vs 7B's 152064) and any Net2Net-expanded FFN width are # picked up automatically -- no stock-config copying. # * lambda_ buffers (ternary fake-quant training machinery) are dropped -- # at inference you run set_lambda(0.0) anyway, so the stored weights ARE # the full-precision weights; the exported model is a plain Qwen2 dense. # * Keys: HF naming passes through; JiRack native naming # (token_emb / blocks.N.* / ffn_w1-w3-w2) is remapped automatically. # # EDIT THE THREE PATHS BELOW. # ============================================================================== import json import os import re import shutil import sys import torch # ========================= EDIT THESE ========================= CKPT_PATH = "model.pt" TOKENIZER_DIR = "." OUTPUT_DIR = "." # rope_theta cannot be inferred from tensor shapes -- set per base model: # DeepSeek-R1-Distill-Qwen-1.5B -> 10000.0 (same as 7B) # DeepSeek-R1-Distill-Qwen-14B -> 1000000.0 # DeepSeek-R1-Distill-Qwen-32B -> 1000000.0 ROPE_THETA = 10000.0 MAX_POSITION = 131072 RMS_NORM_EPS = 1e-6 # Q2_0 = 2-bit ternary {-1, 0, +1} quantization, one fp16 scale per group of # weights -- the real encoding for BitNet-style ternary weights, once inference # actually runs true ternary rather than bf16 dense. For now Q4_K_M remains # the practical choice; the Q2_0 command is just printed ready for later. EMIT_Q2_0_CMD = True Q2_0_GROUP = 64 # 64 = mainline llama.cpp, no fork needed. # ================================================================ # HF Qwen2 key patterns we expect to find (N = layer index) HF_LAYER_KEYS = [ "model.layers.{n}.self_attn.q_proj.weight", "model.layers.{n}.self_attn.q_proj.bias", "model.layers.{n}.self_attn.k_proj.weight", "model.layers.{n}.self_attn.k_proj.bias", "model.layers.{n}.self_attn.v_proj.weight", "model.layers.{n}.self_attn.v_proj.bias", "model.layers.{n}.self_attn.o_proj.weight", "model.layers.{n}.mlp.gate_proj.weight", "model.layers.{n}.mlp.up_proj.weight", "model.layers.{n}.mlp.down_proj.weight", "model.layers.{n}.input_layernorm.weight", "model.layers.{n}.post_attention_layernorm.weight", ] HF_TOP_KEYS = [ "model.embed_tokens.weight", "model.norm.weight", "lm_head.weight", ] def load_state_dict(path): print(f"๐Ÿ“ฅ Loading checkpoint: {path}") ckpt = torch.load(path, map_location="cpu", weights_only=False) sd = ckpt["model"] if isinstance(ckpt, dict) and "model" in ckpt else ckpt if not isinstance(sd, dict): sys.exit("โŒ Checkpoint is not a state_dict and has no 'model' key.") return sd def drop_training_buffers(sd): dropped = [k for k in sd if k.endswith("lambda_")] for k in dropped: del sd[k] if dropped: print(f"๐Ÿงน Dropped {len(dropped)} lambda_ buffers (ternary training machinery).") return sd def normalize_keys(sd): """Pass HF-style keys through; try trivial prefix fixes; else abort with a listing.""" keys = list(sd.keys()) # Case 1: already HF-style if "model.embed_tokens.weight" in sd: print("โœ… Keys already use HF (Qwen2) naming -- no remap needed.") return sd # Case 2: same names but without the leading 'model.' (e.g. 'embed_tokens.weight') if "embed_tokens.weight" in sd: print("๐Ÿ” Keys look HF-like without the 'model.' prefix -- adding it.") out = {} for k, v in sd.items(): if k == "lm_head.weight": out[k] = v else: out["model." + k] = v if "model.embed_tokens.weight" in out: return out # Case 3: JiRack native naming (token_emb / blocks.N.* / ffn_w1-w3-w2) if "token_emb.weight" in sd and any(k.startswith("blocks.") for k in sd): print("๐Ÿ” JiRack native naming detected -- remapping to HF (Qwen2) keys.") hidden = sd["token_emb.weight"].shape[1] block_map = { "norm1.weight": "input_layernorm.weight", "norm2.weight": "post_attention_layernorm.weight", "q_proj.weight": "self_attn.q_proj.weight", "q_proj.bias": "self_attn.q_proj.bias", "k_proj.weight": "self_attn.k_proj.weight", "k_proj.bias": "self_attn.k_proj.bias", "v_proj.weight": "self_attn.v_proj.weight", "v_proj.bias": "self_attn.v_proj.bias", "out_proj.weight": "self_attn.o_proj.weight", "ffn_w1.weight": "mlp.gate_proj.weight", # SwiGLU gate "ffn_w3.weight": "mlp.up_proj.weight", # SwiGLU up "ffn_w2.weight": "mlp.down_proj.weight", # SwiGLU down } out = {"model.embed_tokens.weight": sd["token_emb.weight"]} leftovers = {} blk_pat = re.compile(r"^blocks\.(\d+)\.(.+)$") for k, v in sd.items(): if k == "token_emb.weight": continue m = blk_pat.match(k) if m: idx, sub = m.group(1), m.group(2) if sub == "out_proj.bias": sys.exit("โŒ out_proj has a bias -- Qwen2 arch has no o_proj " "bias, this checkpoint isn't Qwen2-compatible as-is.") if sub not in block_map: sys.exit(f"โŒ Unknown per-block key: {k} -- send this back.") out[f"model.layers.{idx}.{block_map[sub]}"] = v else: leftovers[k] = v # classify the remaining top-level keys by tensor shape for k, v in leftovers.items(): shp = tuple(v.shape) if len(shp) == 1 and shp[0] == hidden: print(f" final norm : {k} -> model.norm.weight") out["model.norm.weight"] = v elif len(shp) == 2 and shp[1] == hidden: print(f" lm head : {k} -> lm_head.weight") out["lm_head.weight"] = v else: sys.exit(f"โŒ Unexplained top-level key: {k} {shp} -- send back.") if "model.norm.weight" not in out: sys.exit("โŒ No final-norm tensor found (1-D, size=hidden). Send the " "full key list (the tail beyond the first 80).") print(f"โœ… Remapped {len(out)} tensors to HF naming.") return out # Case 4: unknown naming -- print everything and stop print("โŒ Unrecognized key naming scheme. Full key list (first 80):") for k in keys[:80]: print(" ", k, tuple(sd[k].shape) if hasattr(sd[k], "shape") else "") print(f" ... total {len(keys)} keys") sys.exit( "\nSend this key list back and I'll add the exact JiRack->HF mapping " "to normalize_keys()." ) def infer_config(sd): """Derive Qwen2 config.json entirely from tensor shapes.""" embed = sd["model.embed_tokens.weight"] vocab_size, hidden_size = embed.shape layer_ids = set() pat = re.compile(r"^model\.layers\.(\d+)\.") for k in sd: m = pat.match(k) if m: layer_ids.add(int(m.group(1))) num_layers = max(layer_ids) + 1 q_w = sd["model.layers.0.self_attn.q_proj.weight"] # [n_heads*head_dim, hidden] k_w = sd["model.layers.0.self_attn.k_proj.weight"] # [n_kv*head_dim, hidden] gate = sd["model.layers.0.mlp.gate_proj.weight"] # [intermediate, hidden] intermediate_size = gate.shape[0] # Qwen2 1.5B/7B/14B/32B all use head_dim=128 (1.5B: 12*128=1536) head_dim = 128 num_attention_heads = q_w.shape[0] // head_dim num_key_value_heads = k_w.shape[0] // head_dim # sanity: every layer's FFN must have the same (expanded) width widths = {sd[f"model.layers.{i}.mlp.gate_proj.weight"].shape[0] for i in layer_ids} if len(widths) != 1: sys.exit(f"โŒ Inconsistent FFN widths across layers: {sorted(widths)}") tie = "lm_head.weight" not in sd cfg = { "architectures": ["Qwen2ForCausalLM"], "model_type": "qwen2", "vocab_size": vocab_size, "hidden_size": hidden_size, "intermediate_size": intermediate_size, "num_hidden_layers": num_layers, "num_attention_heads": num_attention_heads, "num_key_value_heads": num_key_value_heads, "hidden_act": "silu", "max_position_embeddings": MAX_POSITION, "rms_norm_eps": RMS_NORM_EPS, "rope_theta": ROPE_THETA, "tie_word_embeddings": tie, "torch_dtype": "bfloat16", "use_cache": True, "bos_token_id": 151646, "eos_token_id": 151643, } print("๐Ÿงพ Inferred config from tensor shapes:") for k in ("vocab_size", "hidden_size", "intermediate_size", "num_hidden_layers", "num_attention_heads", "num_key_value_heads", "tie_word_embeddings"): print(f" {k} = {cfg[k]}") print(f" rope_theta = {ROPE_THETA} (from the EDIT block -- verify for this base model!)") return cfg def save_hf(sd, cfg): os.makedirs(OUTPUT_DIR, exist_ok=True) device = "cuda" if torch.cuda.is_available() else "cpu" print(f"๐Ÿ”„ Casting weights to bf16 on {device.upper()} ...") for k in sd: t = sd[k] if torch.is_tensor(t) and t.is_floating_point(): sd[k] = t.to(device=device, dtype=torch.bfloat16).cpu().contiguous() try: from safetensors.torch import save_file # single-file safetensors; llama.cpp's converter handles it fine path = os.path.join(OUTPUT_DIR, "model.safetensors") print(f"๐Ÿ’พ Saving {path} ...") save_file(sd, path, metadata={"format": "pt"}) except ImportError: # fallback: pytorch_model.bin, also accepted by convert_hf_to_gguf.py path = os.path.join(OUTPUT_DIR, "pytorch_model.bin") print(f"โš ๏ธ safetensors not installed -- saving {path} instead (also works).") torch.save(sd, path) with open(os.path.join(OUTPUT_DIR, "config.json"), "w") as f: json.dump(cfg, f, indent=2) with open(os.path.join(OUTPUT_DIR, "generation_config.json"), "w") as f: json.dump({"bos_token_id": cfg["bos_token_id"], "eos_token_id": cfg["eos_token_id"], "do_sample": True, "temperature": 0.6, "top_p": 0.95}, f, indent=2) print(f"๐Ÿ“Ž Copying tokenizer from {TOKENIZER_DIR} ...") same_dir = os.path.abspath(TOKENIZER_DIR) == os.path.abspath(OUTPUT_DIR) if same_dir: print(" TOKENIZER_DIR == OUTPUT_DIR -- tokenizer files are already in " "place, skipping copy.") copied = sum( 1 for name in os.listdir(TOKENIZER_DIR) if name.startswith(("tokenizer", "special_tokens", "added_tokens", "vocab", "merges", "chat_template")) ) else: copied = 0 for name in os.listdir(TOKENIZER_DIR): if name.startswith(("tokenizer", "special_tokens", "added_tokens", "vocab", "merges", "chat_template")): shutil.copy2(os.path.join(TOKENIZER_DIR, name), os.path.join(OUTPUT_DIR, name)) copied += 1 if copied == 0: sys.exit(f"โŒ No tokenizer files found in {TOKENIZER_DIR}") print(f" copied {copied} tokenizer files.") def verify(cfg): """Cross-check tokenizer length vs embedding rows.""" try: from transformers import AutoTokenizer tok = AutoTokenizer.from_pretrained(OUTPUT_DIR) n = len(tok) rows = cfg["vocab_size"] if n > rows: sys.exit(f"โŒ Tokenizer has {n} tokens but embedding matrix only {rows} rows -- " f"resize the checkpoint before converting.") print(f"โœ… Tokenizer check: {n} tokens <= {rows} embedding rows " f"({rows - n} spare rows).") except Exception as e: print(f"โš ๏ธ Could not verify tokenizer ({e}) -- continuing anyway.") def main(): if not os.path.exists(CKPT_PATH): sys.exit(f"โŒ {CKPT_PATH} not found") sd = load_state_dict(CKPT_PATH) sd = drop_training_buffers(sd) sd = normalize_keys(sd) cfg = infer_config(sd) save_hf(sd, cfg) verify(cfg) print("\n" + "=" * 78) print("โœ… Stage 1 done. HF model at:", OUTPUT_DIR) print("=" * 78) out_norm = OUTPUT_DIR.rstrip("/") gguf_base = "jirack_1p5b" if out_norm in ("", ".") else out_norm q2_0_block = "" if EMIT_Q2_0_CMD: suffix = "Q2_0" if Q2_0_GROUP == 64 else f"Q2_0_g{Q2_0_GROUP}" fork_note = ( "group-64 is in mainline llama.cpp -- no fork needed, CPU/Metal ready." if Q2_0_GROUP == 64 else "group-128 needs a CUDA fork -- not needed on CPU-only." ) q2_0_block = """ Ternary quantization (Q2_0, 2 bits/weight, {{-1,0,+1}} + fp16 group scale -- this is the real encoding for BitNet-style ternary weights, once your model actually runs true ternary at inference rather than bf16 dense): {fork_note} ./build/bin/llama-quantize {gguf} {gguf_q2} {suffix} """.format( fork_note=fork_note, gguf=gguf_base + ".gguf", gguf_q2=gguf_base + f".{suffix}.gguf", suffix=suffix, ) print(""" Stage 2 -- make the GGUF (one-time llama.cpp setup, then per model): git clone https://github.com/ggml-org/llama.cpp /mnt/nfs_share/llama.cpp cd /mnt/nfs_share/llama.cpp pip install -r requirements.txt python convert_hf_to_gguf.py {out} \\ --outfile {gguf} --outtype bf16 Optional dense quantization (build llama.cpp first: cmake -B build && cmake --build build -j): ./build/bin/llama-quantize {gguf} {gguf_q} Q4_K_M {q2_0_block}""".format( out=OUTPUT_DIR, gguf=gguf_base + ".gguf", gguf_q=gguf_base + ".Q4_K_M.gguf", q2_0_block=q2_0_block, )) if __name__ == "__main__": main()