From 42aa1b5282ec5132bf7665a4dbcd890ce94bfdc8 Mon Sep 17 00:00:00 2001 From: ModelHub XC Date: Tue, 25 Aug 2026 05:59:19 +0800 Subject: [PATCH] =?UTF-8?q?=E5=88=9D=E5=A7=8B=E5=8C=96=E9=A1=B9=E7=9B=AE?= =?UTF-8?q?=EF=BC=8C=E7=94=B1ModelHub=20XC=E7=A4=BE=E5=8C=BA=E6=8F=90?= =?UTF-8?q?=E4=BE=9B=E6=A8=A1=E5=9E=8B?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Model: CMSManhattan/JiRackUltra_1b Source: Original Platform --- .gitattributes | 40 ++++ JiRackTernaryUltra_1b.py | 469 +++++++++++++++++++++++++++++++++++++ JiRackUltra_1b.gguf | 3 + JiRackUltra_1b_Q3_K_M.gguf | 3 + JiRackUltra_1b_Q4_K_M.gguf | 3 + NOTICE.md | 7 + README.md | 218 +++++++++++++++++ chat_jirack_1b.py | 183 +++++++++++++++ chat_template.jinja | 1 + config.json | 21 ++ generation_config.json | 7 + get_tool_call.py | 229 ++++++++++++++++++ gguf.txt | 14 ++ gguf_chat.sh | 7 + jirack_to_gguf_1p5b.py | 371 +++++++++++++++++++++++++++++ model.pt | 3 + model.safetensors | 3 + quant.sh | 5 + tokenizer.json | 3 + tokenizer_config.json | 131 +++++++++++ 20 files changed, 1721 insertions(+) create mode 100644 .gitattributes create mode 100644 JiRackTernaryUltra_1b.py create mode 100644 JiRackUltra_1b.gguf create mode 100644 JiRackUltra_1b_Q3_K_M.gguf create mode 100644 JiRackUltra_1b_Q4_K_M.gguf create mode 100644 NOTICE.md create mode 100644 README.md create mode 100644 chat_jirack_1b.py create mode 100644 chat_template.jinja create mode 100644 config.json create mode 100644 generation_config.json create mode 100644 get_tool_call.py create mode 100644 gguf.txt create mode 100644 gguf_chat.sh create mode 100644 jirack_to_gguf_1p5b.py create mode 100644 model.pt create mode 100644 model.safetensors create mode 100644 quant.sh create mode 100644 tokenizer.json create mode 100644 tokenizer_config.json diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..a84c6d2 --- /dev/null +++ b/.gitattributes @@ -0,0 +1,40 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text +Docker/web/jirack.apk filter=lfs diff=lfs merge=lfs -text +JiRackUltra_1b.gguf filter=lfs diff=lfs merge=lfs -text +JiRackUltra_1b_Q3_K_M.gguf filter=lfs diff=lfs merge=lfs -text +JiRackUltra_1b_Q4_K_M.gguf filter=lfs diff=lfs merge=lfs -text +tokenizer.json filter=lfs diff=lfs merge=lfs -text diff --git a/JiRackTernaryUltra_1b.py b/JiRackTernaryUltra_1b.py new file mode 100644 index 0000000..c85e6c9 --- /dev/null +++ b/JiRackTernaryUltra_1b.py @@ -0,0 +1,469 @@ +#%%writefile JiRackTernaryUltra_1p5b.py +# ============================================================================= +# COPYRIGHT © 2026 Konstantin Vladimirovich Grabko. ALL RIGHTS RESERVED. +# JiRack Ultra Ternary Transformer +# +# CMS Manhattan JiRack Technology — PATENT PENDING +# +# This code is proprietary. +# Personal and non-commercial research use is allowed. +# Any commercial use, derivative works for profit, or distribution +# requires a paid license and 5% royalty. +# +# Unauthorized commercial use is strictly prohibited. +# Contact: grabko@cmsmanhattan.com +# ============================================================================= +import math + +import torch +import torch.nn as nn +import torch.nn.functional as F +from torch.utils.checkpoint import checkpoint + +# ==================== CONFIG CONSTANTS [DS1.5-1] ==================== +VOCAB_SIZE = 151936 +HIDDEN_SIZE = 1536 +INTERMEDIATE_SIZE = 8960 +NUM_LAYERS = 28 +NUM_HEADS = 12 +NUM_KV_HEADS = 2 +HEAD_DIM = 128 # 12 * 128 = 1536 = hidden (q); kv dim = 2*128 = 256 +MAX_SEQ_LEN = 4096 # [DS-6] raise for long-context (ckpt supports 131072) +ROPE_THETA = 10000.0 # Qwen2.5-Math value (NOT Llama-3's 500000) +RMS_EPS = 1e-6 # Qwen2 value (NOT Llama-3's 1e-5) +ROPE_SCALE_FACTOR = 1.0 +INIT_STD = 0.02 +ATTN_QKV_BIAS = True # [DS-2] Qwen2: bias on q/k/v only +# ================================================================= + + +# [FIX-6] Feature-detect native GQA support in SDPA (PyTorch >= 2.5). +def _detect_sdpa_gqa() -> bool: + try: + q = torch.zeros(1, 2, 1, 8) + kv = torch.zeros(1, 1, 1, 8) + F.scaled_dot_product_attention(q, kv, kv, enable_gqa=True) + return True + except TypeError: + return False + except Exception: + return False + +_SDPA_HAS_GQA = _detect_sdpa_gqa() + + +class JiRackConfig: + def __init__(self): + self.vocab_size = VOCAB_SIZE + self.hidden_size = HIDDEN_SIZE + self.intermediate_size = INTERMEDIATE_SIZE + self.num_hidden_layers = NUM_LAYERS + self.num_attention_heads = NUM_HEADS + self.num_key_value_heads = NUM_KV_HEADS + self.head_dim = HEAD_DIM + self.max_seq_len = MAX_SEQ_LEN + self.rope_theta = ROPE_THETA + self.rms_norm_eps = RMS_EPS + self.rope_scale_factor = ROPE_SCALE_FACTOR + self.init_std = INIT_STD + self.attn_qkv_bias = ATTN_QKV_BIAS + + +# ==================== RoPE — HALF-SPLIT (HF convention) [DS-3] ==================== +def precompute_freqs_cis( + dim: int, + end: int, + theta: float = ROPE_THETA, + scale_factor: float = ROPE_SCALE_FACTOR, +): + """cos/sin of shape (end, dim), HF half-split layout: the (dim/2) + frequency vector is CONCATENATED with itself (torch.cat), not + interleaved. Matches transformers' LlamaRotaryEmbedding/Qwen2.""" + freqs = 1.0 / (theta ** (torch.arange(0, dim, 2).float() / dim)) + if scale_factor > 1.0: + freqs = freqs / scale_factor + t = torch.arange(end, dtype=torch.float32) + freqs = torch.outer(t, freqs) # (end, dim/2) + emb = torch.cat((freqs, freqs), dim=-1) # (end, dim) — half-split + return torch.cos(emb), torch.sin(emb) + + +def rotate_half(x): + """HF convention: (-x2, x1) where x1/x2 are the two HALVES of head_dim.""" + x1 = x[..., : x.shape[-1] // 2] + x2 = x[..., x.shape[-1] // 2:] + return torch.cat((-x2, x1), dim=-1) + + +def apply_rotary_emb(xq, xk, cos, sin): + """Half-split RoPE, identical math to transformers.apply_rotary_pos_emb. + cos/sin: (T, head_dim); q/k: (B, H, T, head_dim).""" + cos = cos[None, None, :, :] + sin = sin[None, None, :, :] + xq_out = (xq * cos) + (rotate_half(xq) * sin) + xk_out = (xk * cos) + (rotate_half(xk) * sin) + return xq_out, xk_out + + +class BitLinear(nn.Linear): + """BitNet b1.58-style fake-quant linear with lambda warmup. + Identical to the 10B version ([FIX-1..5] preserved); bias — when + present ([DS-2]) — stays full precision ([DS-5]).""" + + def __init__(self, in_features, out_features, bias=False): + super().__init__(in_features, out_features, bias=bias) + self.eps = 1e-5 + # [FIX-4] Buffer -> saved in state_dict, survives checkpoint resume. + self.register_buffer("lambda_", torch.zeros(()), persistent=True) + + def forward(self, x: torch.Tensor) -> torch.Tensor: + # Fast path (exact at lambda=0 by continuity). + if not self.training and float(self.lambda_) < 1e-6: + return F.linear(x, self.weight, self.bias) + + lam = self.lambda_.to(x.dtype) + + # === Weights: per-tensor absmean ternary (b1.58) === + w = self.weight + gamma = w.float().abs().mean().clamp(min=self.eps).to(w.dtype) # [FIX-5] + w_quant = torch.clamp(torch.round(w / gamma), -1.0, 1.0) * gamma + w_effective = w + lam * (w_quant - w).detach() + + # === Activations: per-token absmax int8 ([FIX-2],[FIX-3]) === + x_scale = 127.0 / x.abs().max(dim=-1, keepdim=True).values.clamp(min=self.eps) + x_quant = torch.clamp(torch.round(x * x_scale), -128.0, 127.0) / x_scale + x_effective = x + lam * (x_quant - x).detach() + + # [FIX-1] Dequantized operands -> no post-matmul rescale. + # [DS-5] bias added in full precision by F.linear. + return F.linear(x_effective, w_effective, self.bias) + + +class RMSNorm(nn.Module): + def __init__(self, dim, eps=RMS_EPS): + super().__init__() + self.eps = eps + self.weight = nn.Parameter(torch.ones(dim)) + + def forward(self, x): + # [FIX-5] Compute statistics in fp32, cast back to input dtype. + dtype = x.dtype + x = x.float() + x = x * torch.rsqrt(x.pow(2).mean(-1, keepdim=True) + self.eps) + return (x * self.weight.float()).to(dtype) + + +class TransformerBlock(nn.Module): + def __init__(self, config, use_checkpoint=False): + super().__init__() + self.use_checkpoint = use_checkpoint + self.n_heads = config.num_attention_heads + self.n_kv_heads = config.num_key_value_heads + self.head_dim = config.head_dim + self.n_rep = self.n_heads // self.n_kv_heads + + self.norm1 = RMSNorm(config.hidden_size, eps=config.rms_norm_eps) + self.norm2 = RMSNorm(config.hidden_size, eps=config.rms_norm_eps) + + qkv_bias = config.attn_qkv_bias # [DS-2] + self.q_proj = BitLinear(config.hidden_size, + self.n_heads * self.head_dim, bias=qkv_bias) + self.k_proj = BitLinear(config.hidden_size, + self.n_kv_heads * self.head_dim, bias=qkv_bias) + self.v_proj = BitLinear(config.hidden_size, + self.n_kv_heads * self.head_dim, bias=qkv_bias) + self.out_proj = BitLinear(self.n_heads * self.head_dim, + config.hidden_size, bias=False) + + self.ffn_w1 = BitLinear(config.hidden_size, config.intermediate_size, bias=False) # gate + self.ffn_w3 = BitLinear(config.hidden_size, config.intermediate_size, bias=False) # up + self.ffn_w2 = BitLinear(config.intermediate_size, config.hidden_size, bias=False) # down + + def forward(self, x, freqs_cos, freqs_sin): + if self.use_checkpoint and self.training: + return checkpoint( + self._forward_impl, x, freqs_cos, freqs_sin, use_reentrant=False + ) + return self._forward_impl(x, freqs_cos, freqs_sin) + + def _forward_impl(self, x, freqs_cos, freqs_sin): + h = self.norm1(x) + B, T, _ = h.shape + + q = self.q_proj(h).view(B, T, self.n_heads, self.head_dim).transpose(1, 2) + k = self.k_proj(h).view(B, T, self.n_kv_heads, self.head_dim).transpose(1, 2) + v = self.v_proj(h).view(B, T, self.n_kv_heads, self.head_dim).transpose(1, 2) + + q, k = apply_rotary_emb(q, k, freqs_cos, freqs_sin) # [DS-3] half-split + + if self.n_rep > 1 and _SDPA_HAS_GQA: # [FIX-6] + attn_out = F.scaled_dot_product_attention( + q, k, v, is_causal=True, enable_gqa=True + ) + else: + if self.n_rep > 1: + k = k.repeat_interleave(self.n_rep, dim=1) + v = v.repeat_interleave(self.n_rep, dim=1) + attn_out = F.scaled_dot_product_attention(q, k, v, is_causal=True) + + attn_out = attn_out.transpose(1, 2).contiguous().view(B, T, -1) + + x = x + self.out_proj(attn_out) + + m = self.norm2(x) + gate = F.silu(self.ffn_w1(m)) + up = self.ffn_w3(m) + x = x + self.ffn_w2(gate * up) + + return x + + +class JiRackTransformer(nn.Module): + def __init__(self, config: JiRackConfig = None, use_checkpoint=False): + super().__init__() + self.config = config if config is not None else JiRackConfig() + self.use_checkpoint = use_checkpoint + + self.token_emb = nn.Embedding(self.config.vocab_size, self.config.hidden_size) + self.blocks = nn.ModuleList([ + TransformerBlock(self.config, self.use_checkpoint) + for _ in range(self.config.num_hidden_layers) + ]) + self.ln_f = RMSNorm(self.config.hidden_size, eps=self.config.rms_norm_eps) + # [DS1.5-2] tie_word_embeddings = False (verified) -> separate lm_head. + self.lm_head = nn.Linear(self.config.hidden_size, self.config.vocab_size, bias=False) + + cos, sin = precompute_freqs_cis( + dim=self.config.head_dim, + end=self.config.max_seq_len, + theta=self.config.rope_theta, + scale_factor=self.config.rope_scale_factor, + ) + self.register_buffer("freqs_cos", cos, persistent=False) + self.register_buffer("freqs_sin", sin, persistent=False) + + # [FIX-8] Only relevant when training from scratch; harmless before + # load_hf_state_dict() overwrites everything. + self._init_weights() + + def _init_weights(self): + std = self.config.init_std + resid_std = std / math.sqrt(2 * self.config.num_hidden_layers) + + nn.init.normal_(self.token_emb.weight, mean=0.0, std=std) + nn.init.normal_(self.lm_head.weight, mean=0.0, std=std) + + for block in self.blocks: + for lin in (block.q_proj, block.k_proj, block.v_proj, + block.ffn_w1, block.ffn_w3): + nn.init.normal_(lin.weight, mean=0.0, std=std) + if lin.bias is not None: + nn.init.zeros_(lin.bias) + for lin in (block.out_proj, block.ffn_w2): + nn.init.normal_(lin.weight, mean=0.0, std=resid_std) + if lin.bias is not None: + nn.init.zeros_(lin.bias) + + # ---------------- lambda warmup hooks (unchanged) ---------------- + def set_lambda(self, lambda_value: float): + for module in self.modules(): + if isinstance(module, BitLinear): + module.lambda_.fill_(lambda_value) + + def get_lambda(self) -> float: + for module in self.modules(): + if isinstance(module, BitLinear): + return float(module.lambda_) + return 0.0 + + def forward(self, input_ids): + seq_len = input_ids.shape[1] + x = self.token_emb(input_ids) + + cos = self.freqs_cos[:seq_len].to(device=x.device, dtype=x.dtype) + sin = self.freqs_sin[:seq_len].to(device=x.device, dtype=x.dtype) + + for block in self.blocks: + x = block(x, cos, sin) + + return self.lm_head(self.ln_f(x)) + + # ------------------------------------------------------------------ + # [DS-4] HF Qwen2 -> JiRack weight mapping. + # Usage: + # from transformers import AutoModelForCausalLM + # hf = AutoModelForCausalLM.from_pretrained( + # "deepseek-ai/DeepSeek-R1-Distill-Qwen-1.5B", torch_dtype=torch.float32) + # model.load_hf_state_dict(hf.state_dict()) + # or load safetensors shards directly and merge them into one dict. + # No RoPE permutation is needed: this model now uses the same + # half-split rotation as HF ([DS-3]). + # ------------------------------------------------------------------ + @torch.no_grad() + def load_hf_state_dict(self, hf_sd: dict, strict: bool = True): + mapped = {} + mapped["token_emb.weight"] = hf_sd["model.embed_tokens.weight"] + mapped["ln_f.weight"] = hf_sd["model.norm.weight"] + if "lm_head.weight" in hf_sd: + mapped["lm_head.weight"] = hf_sd["lm_head.weight"] + else: + # fallback for third-party re-uploads that strip lm_head + # (official 1.5B distill DOES ship lm_head.weight — untied) + mapped["lm_head.weight"] = hf_sd["model.embed_tokens.weight"] + + for i in range(self.config.num_hidden_layers): + hf = f"model.layers.{i}" + jr = f"blocks.{i}" + mapped[f"{jr}.norm1.weight"] = hf_sd[f"{hf}.input_layernorm.weight"] + mapped[f"{jr}.norm2.weight"] = hf_sd[f"{hf}.post_attention_layernorm.weight"] + + mapped[f"{jr}.q_proj.weight"] = hf_sd[f"{hf}.self_attn.q_proj.weight"] + mapped[f"{jr}.k_proj.weight"] = hf_sd[f"{hf}.self_attn.k_proj.weight"] + mapped[f"{jr}.v_proj.weight"] = hf_sd[f"{hf}.self_attn.v_proj.weight"] + mapped[f"{jr}.q_proj.bias"] = hf_sd[f"{hf}.self_attn.q_proj.bias"] + mapped[f"{jr}.k_proj.bias"] = hf_sd[f"{hf}.self_attn.k_proj.bias"] + mapped[f"{jr}.v_proj.bias"] = hf_sd[f"{hf}.self_attn.v_proj.bias"] + mapped[f"{jr}.out_proj.weight"] = hf_sd[f"{hf}.self_attn.o_proj.weight"] + + mapped[f"{jr}.ffn_w1.weight"] = hf_sd[f"{hf}.mlp.gate_proj.weight"] + mapped[f"{jr}.ffn_w3.weight"] = hf_sd[f"{hf}.mlp.up_proj.weight"] + mapped[f"{jr}.ffn_w2.weight"] = hf_sd[f"{hf}.mlp.down_proj.weight"] + + missing, unexpected = self.load_state_dict(mapped, strict=False) + # lambda_ buffers are OURS (not in HF) — they legitimately stay missing. + real_missing = [k for k in missing if not k.endswith("lambda_")] + if strict: + assert not real_missing, f"missing from HF checkpoint: {real_missing}" + assert not unexpected, f"unexpected keys: {unexpected}" + print(f"✅ HF weights loaded: {len(mapped)} tensors " + f"({len(real_missing)} missing, {len(unexpected)} unexpected)") + return real_missing, unexpected + + # ------------------------------------------------------------------ + # [FIX-9] Ternary export (biases included, full precision, [DS-5]). + # ------------------------------------------------------------------ + @torch.no_grad() + def export_ternary_state_dict(self): + out = {} + for name, module in self.named_modules(): + if isinstance(module, BitLinear): + w = module.weight.float() + gamma = w.abs().mean().clamp(min=module.eps) + codes = torch.clamp(torch.round(w / gamma), -1.0, 1.0).to(torch.int8) + out[f"{name}.codes"] = codes + out[f"{name}.gamma"] = gamma + if module.bias is not None: + out[f"{name}.bias"] = module.bias.detach().clone() + out["token_emb.weight"] = self.token_emb.weight.detach().clone() + out["lm_head.weight"] = self.lm_head.weight.detach().clone() + out["ln_f.weight"] = self.ln_f.weight.detach().clone() + for name, module in self.named_modules(): + if isinstance(module, RMSNorm) and name != "ln_f": + out[f"{name}.weight"] = module.weight.detach().clone() + return out + + +# Convenience aliases so existing training scripts barely change: +JiRackConfig = JiRackConfig # drop-in name compat (optional) +JiRackTransformer = JiRackTransformer +JiRackConfig = JiRackConfig # lets ds7b-style imports work +JiRackTransformer = JiRackTransformer + + +# ============================================================================= +# Smoke test: python JiRackTernaryPyTorch_ds1p5b.py (tiny config, CPU, seconds) +# ============================================================================= +if __name__ == "__main__": + class TinyConfig(JiRackConfig): + def __init__(self): + super().__init__() + self.vocab_size = 256 + self.hidden_size = 64 + self.intermediate_size = 128 + self.num_hidden_layers = 2 + self.num_attention_heads = 4 + self.num_key_value_heads = 2 + self.head_dim = 16 + self.max_seq_len = 64 + + torch.manual_seed(0) + model = JiRackTransformer(TinyConfig()).eval() + ids = torch.randint(0, 256, (2, 32)) + + with torch.no_grad(): + model.set_lambda(0.0) + y0 = model(ids) + model.set_lambda(1e-4) + y_eps = model(ids) + model.set_lambda(1.0) + y1 = model(ids) + + # 1) Continuity in lambda. + rel_jump = (y_eps - y0).norm() / y0.norm() + print(f"relative change at lambda=1e-4: {rel_jump.item():.2e} (must be ~1e-4)") + assert rel_jump < 1e-2, "lambda warmup is not continuous!" + + # 2) Output scale sanity at full quantization. + ratio = y1.std() / y0.std() + print(f"std ratio lambda=1 vs lambda=0: {ratio.item():.3f} (must be O(1))") + assert 0.1 < ratio.item() < 10.0, "output scale collapsed or exploded!" + + # 3) Gradients flow through STE at lambda=1 (weights AND qkv biases). + model.train() + model.set_lambda(1.0) + loss = model(ids).float().pow(2).mean() + loss.backward() + g = model.blocks[0].q_proj.weight.grad + gb = model.blocks[0].q_proj.bias.grad + assert g is not None and torch.isfinite(g).all() and g.abs().sum() > 0 + assert gb is not None and torch.isfinite(gb).all(), "qkv bias got no grad!" + print(f"grad norms q_proj: weight={g.norm().item():.4f}, bias={gb.norm().item():.4f}") + + # 4) lambda survives a state_dict round-trip. + sd = model.state_dict() + model2 = JiRackTransformerDS1p5B(TinyConfig()) + model2.load_state_dict(sd) + assert abs(model2.get_lambda() - 1.0) < 1e-9, "lambda_ not serialized!" + print("lambda serialization: OK") + + # 5) HF name mapping round-trip on the tiny config: build a fake HF + # dict from our own weights, load it back, outputs must match. + fake_hf = { + "model.embed_tokens.weight": model.token_emb.weight.detach().clone(), + "model.norm.weight": model.ln_f.weight.detach().clone(), + "lm_head.weight": model.lm_head.weight.detach().clone(), + } + for i, blk in enumerate(model.blocks): + p = f"model.layers.{i}" + fake_hf[f"{p}.input_layernorm.weight"] = blk.norm1.weight.detach().clone() + fake_hf[f"{p}.post_attention_layernorm.weight"] = blk.norm2.weight.detach().clone() + fake_hf[f"{p}.self_attn.q_proj.weight"] = blk.q_proj.weight.detach().clone() + fake_hf[f"{p}.self_attn.k_proj.weight"] = blk.k_proj.weight.detach().clone() + fake_hf[f"{p}.self_attn.v_proj.weight"] = blk.v_proj.weight.detach().clone() + fake_hf[f"{p}.self_attn.q_proj.bias"] = blk.q_proj.bias.detach().clone() + fake_hf[f"{p}.self_attn.k_proj.bias"] = blk.k_proj.bias.detach().clone() + fake_hf[f"{p}.self_attn.v_proj.bias"] = blk.v_proj.bias.detach().clone() + fake_hf[f"{p}.self_attn.o_proj.weight"] = blk.out_proj.weight.detach().clone() + fake_hf[f"{p}.mlp.gate_proj.weight"] = blk.ffn_w1.weight.detach().clone() + fake_hf[f"{p}.mlp.up_proj.weight"] = blk.ffn_w3.weight.detach().clone() + fake_hf[f"{p}.mlp.down_proj.weight"] = blk.ffn_w2.weight.detach().clone() + + model3 = JiRackTransformer(TinyConfig()).eval() + model3.load_hf_state_dict(fake_hf) + model3.set_lambda(0.0) + model.eval(); model.set_lambda(0.0) + with torch.no_grad(): + y_ref = model(ids) + y_map = model3(ids) + assert torch.allclose(y_ref, y_map, atol=1e-5), "HF mapping mismatch!" + print("HF name-mapping round-trip: OK") + + # 6) Export produces genuinely ternary codes + preserved biases. + exported = model.export_ternary_state_dict() + codes = exported["blocks.0.q_proj.codes"] + assert set(codes.unique().tolist()) <= {-1, 0, 1} + assert "blocks.0.q_proj.bias" in exported, "qkv bias lost in export!" + print(f"export: {sum('codes' in k for k in exported)} ternary tensors, " + f"biases preserved, OK") + + print("\nAll smoke tests passed.") diff --git a/JiRackUltra_1b.gguf b/JiRackUltra_1b.gguf new file mode 100644 index 0000000..a4fc68b --- /dev/null +++ b/JiRackUltra_1b.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:19bb919b2deb433fba4b1116d857678def467b49ca881b5f9e6dc6ad27664744 +size 3560416160 diff --git a/JiRackUltra_1b_Q3_K_M.gguf b/JiRackUltra_1b_Q3_K_M.gguf new file mode 100644 index 0000000..a1d5c54 --- /dev/null +++ b/JiRackUltra_1b_Q3_K_M.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:73cc639081a63a3793448991c1c1ca05ea82e5a9f65736b9e571f32aca10f31d +size 924455840 diff --git a/JiRackUltra_1b_Q4_K_M.gguf b/JiRackUltra_1b_Q4_K_M.gguf new file mode 100644 index 0000000..389c692 --- /dev/null +++ b/JiRackUltra_1b_Q4_K_M.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8db8cb25c578442e4a05a664e02c5ec90dda393671b9ef2ca02d37170bb0335f +size 1117320608 diff --git a/NOTICE.md b/NOTICE.md new file mode 100644 index 0000000..346349b --- /dev/null +++ b/NOTICE.md @@ -0,0 +1,7 @@ +- QWEN 2.5 - Apache 2.0 license +- DEEPSEEK R1 - MIT license +- JIRACK JIPRECISION TOKENIZER - CMS Manhattan license +- PATENT KONSTANTIN GRABKO +- JIRACK TERNARY ARCHITECTURE - CMS Manhattan license +- JIRACK WEB UI - CMS Manhattan license +- TOOL ACE - Apache 2.0 license \ No newline at end of file diff --git a/README.md b/README.md new file mode 100644 index 0000000..8f8021e --- /dev/null +++ b/README.md @@ -0,0 +1,218 @@ +--- +language: +- en +- zh +- ja +- ko +- fr +- es +- pt +- de +- it +- ru +- ar +- vi +- th +tags: +- text-generation +- ternary +- bitnet +- 1.58bit +- cpu +- gguf +- qwen2.5 +- deepseek +- efficient +- low-memory +- jirack +- web-ui +- routing +- tool-call +- robotics +license: mit +--- +# JiRack Ultra 1B (CPU) +A fast and efficient ~1.5B model optimized for CPU inference. The model was refactored with BitNet features and an updated tokenizer that includes new **Routing**, **Tool call**, and **Robotics** tags. Built on a redesigned DeepSeek R1 architecture with native ternary (BitNet-style) support and ready-to-run GGUF quantizations. +- JiRack is a cloud-ready model that helps save money on cloud infrastructure. It can be used as an expert model in RAG deployments, with the ONNX JiRack Java server as an alternative. +- Subscription: **$1 per month per user** (updated license for non-company use). +- Corp Subscription: **$3 per month per user** (updated license for company use). +- It works without subscription but send message about subscription + +# Ollama production support +- We are working to support JiRack on Ollama for production systems also +- added Jirack chat without reasoning feature https://ollama.com/cmsmanhattan +- Follow fresh Ollama platform updates + +# JiRack sevice options +- Current quantizations were done from the FP16 model, but the model allows for more compression thanks to its ternary architecture. +- If you need to do ternary compression, please write to me and I'll perform QAT from your dataset, tailored specifically to your task. +- Plus double QAT via ONNX QAT. +- Adapt train process to avoid catastrophic forgetting with NDA +- Adapt train process to avoid fast plato in training with NDA +- Convert model to TQ2_0 with support AVX2 and AVX-512 CPU instructions for high performance on CPU +- Adapts to agentic or instruct models for tool calling, using the JiRak tokenizer to enable high-quality tool calling on small models — built as a domain-specific tool expert. +- Deployment and scale + + + +# Spring Boot AI tool calls examples for JiRack Ultra series +- Tool call library on java for Enterprise https://github.com/alibaba/spring-ai-alibaba + +# GoEx AI tool calls examples for JiRack Ultra series +- Tool call library on python https://github.com/ShishirPatil/gorilla + +# JiRack Ultra 1 tool calls to boost tool call quality +- Use JiRack Precision tokenzer tags for tool calls with ToolBench https://github.com/OpenBMB/ToolBench +- https://huggingface.co/xalss/Qwen2-7B-Instruct-glaive-function-calling +- https://huggingface.co/datasets/NousResearch/hermes-function-calling-v1 +- Add JiRack tool call tags in the dataset and modify tool call processor if needed + + + + +# JiRack RoboTech +- Advanced Tokenizer with Robotics & Routing & Tool calls Tokenizer and other +- [CMSManhattan/JiRackPrecisionTokenizer](https://huggingface.co/CMSManhattan/JiRackPrecisionTokenizer) + + + +## Available Variants +| Tag | Quant | Size | Approx. RAM | Description | +|-----|-------|------|-------------|-------------| +| `cmsmanhattan/jirack-ultra-1b-cpu:latest` | Full | 0.55 GB | ~1.8 GB | Full ternary reference | +| `cmsmanhattan/jirack-ultra-1b-cpu-q4:latest` | Q4_K_M | 0.38 GB | ~1.4 GB | Recommended balance | +| `cmsmanhattan/jirack-ultra-1b-cpu-q3:latest` | Q3_K_M | 0.31 GB | ~1.2 GB | Good quality / size trade-off | +| `cmsmanhattan/jirack-ultra-1b-cpu-q2:latest` | Q2_K | 0.24 GB | ~1.0 GB | Maximum compression | +## Quick Start +### Run with Docker +**Default CPU (Q4 recommended)** +```bash +docker run -d \ + --name jirack_ultra_1b \ + -p 7869:7869 \ + --cpus=16 \ + -e THREADS=16 \ + -e THREADS_BATCH=16 \ + --restart unless-stopped \ + cmsmanhattan/jirack-ultra-1b-cpu-q4:latest +``` +**Q3** +```bash +docker run -d \ + --name jirack_ultra_1b \ + -p 7869:7869 \ + --cpus=16 \ + -e THREADS=16 \ + -e THREADS_BATCH=16 \ + --restart unless-stopped \ + cmsmanhattan/jirack-ultra-1b-cpu-q3:latest +``` +**Q2 (lowest memory)** +```bash +docker run -d \ + --name jirack_ultra_1b \ + -p 7869:7869 \ + --cpus=16 \ + -e THREADS=16 \ + -e THREADS_BATCH=16 \ + --restart unless-stopped \ + cmsmanhattan/jirack-ultra-1b-cpu-q2:latest +``` +**Full precision** +```bash +docker run -d \ + --name jirack_ultra_1b \ + -p 7869:7869 \ + --cpus=16 \ + -e THREADS=16 \ + -e THREADS_BATCH=16 \ + --restart unless-stopped \ + cmsmanhattan/jirack-ultra-1b-cpu:latest +``` +**Multi CPU** +```bash +docker run -d \ + --name jirack_ultra_1b \ + -p 7869:7869 \ + --cpus=16 \ + -e THREADS=16 \ + -e THREADS_BATCH=16 \ + --restart unless-stopped \ + --memory=4g \ + --cpus=4 \ + cmsmanhattan/jirack-ultra-1b-cpu-q4:latest +``` +### Docker Compose Example +```yaml +services: + jirack: + image: cmsmanhattan/jirack-ultra-1b-cpu-q4:latest + container_name: jirack_ultra_1b + ports: + - "7869:7869" + volumes: + - .:/app + - ./web:/app/web + environment: + - MAX_TOKENS=2048 + - TEMPERATURE=0.7 + - TOP_P=0.9 + - DEFAULT_STREAM=False + - INTRA_THREADS=4 + - USE_ENV_ALLOCATOR=1 + - THREADS=16 + - THREADS_BATCH=16 + deploy: + resources: + limits: + memory: 4g +``` +## Access the UI +Once the container is running, open your browser and navigate to: +`http://localhost:7869` +This opens the JiRack UI — a clean web interface. +## Changing the Port +The listening port can be easily modified directly from the **Settings** panel within the JiRack UI. +## Licensing +- The JiRack Ultra 1B model is provided under a commercial license ($12 per user per year). +- All JiRack UI clients are provided under a commercial license. +- However, the UI clients can be used for free when running together with the official JiRack Docker containers, as long as they are not redistributed separately. +For commercial licensing, cluster deployment, or enterprise use of JiRack models, please contact us. +- **JiRack MS Windows 11 Desktop Client (with Ollama API):** + https://huggingface.co/kgrabko/JiRackTernary_1b/resolve/main/jirack-chat.zip +- **Live email chat with the model:** support@cmsmanhattan.com +## Hardware Recommendations +### Recommended Hardware for JiRack Ultra 1B (single Docker container) +| Use Case | CPU | RAM | Recommended Quant | Expected Speed | Recommendation | +|-------------------|------------------------------|----------|-------------------|---------------------|----------------| +| Recommended | Ryzen 5 / Intel i5 | 4–8 GB | Q4_K_M | Excellent interactive | Best choice | +| High Performance | Ryzen 7 / Intel i7 | 8–16 GB | Full / Q4 | Excellent | Excellent | +| Low Memory | Modern 4+ core CPU | 2–4 GB | Q3_K_M or Q2_K | Usable | Acceptable | +| Edge / Minimal | Laptop / SBC CPU | 2 GB | Q2_K | Acceptable | Budget option | +## Important Memory Notes +Even though the quantized 1B models are very small, we recommend the following for best experience: +- Q4_K_M: 2–4 GB system RAM minimum +- Q3_K_M / Q2_K: 1.5–3 GB system RAM +- Full precision: 3–4 GB+ system RAM recommended +Reasons for extra headroom: +- KV-cache consumption during generation +- Runtime overhead and temporary buffers +- System stability and avoiding out-of-memory errors +- Room for larger context windows +**Minimum recommended (Q4):** 2–3 GB system RAM +**Ideal:** 4–8 GB system RAM +I added the default model in full precision. This serves as the base for quantization, allowing us to find the optimal balance between model size and performance. +## Architecture Notes +- **Refactored with BitNet features**: Native BitLinear ternary path (b1.58-style) with λ-warmup STE +- **Updated tokenizer**: Extended with new special tags for **Routing**, **Tool call**, and **Robotics** +- Base: Redesigned Llama-3.2-1B style (Hidden 2048, Intermediate 8192, 16 layers, GQA 32/8, vocab 128256) +- RoPE θ = 10000, RMSNorm ε = 1e-6 +- Ready-to-run GGUF quantizations (Q2_K, Q3_K_M, Q4_K_M) +## 📧 Contact & Licensing +For joint venture opportunities, hardware integration, or licensing inquiries: +- **Email:** grabko@cmsmanhattan.com +- **Phone:** +1 (516) 777-0945 +- **Location:** New York, USA + +## License +MIT License \ No newline at end of file diff --git a/chat_jirack_1b.py b/chat_jirack_1b.py new file mode 100644 index 0000000..30e3ed5 --- /dev/null +++ b/chat_jirack_1b.py @@ -0,0 +1,183 @@ +# ============================================================================== +# JiRack 32B Chat (DeepSeek-R1-Distill-Qwen-32B edition, extended tokenizer) +# COPYRIGHT (c) 2026 Konstantin Vladimirovich Grabko. +# +# Mirrors chat_jirack_7b.py, adjusted for the 32B checkpoint: +# * VOCAB_SIZE=152064, hidden=5120 (per your verified checkpoint shapes) +# * Remember the MKL SIMD dispatch fix if you ever run this on CPU: +# export MKL_ENABLE_INSTRUCTIONS=AVX +# export MKL_DEBUG_CPU_TYPE=5 +# (this CPU only exposes AVX, no AVX2/AVX512 -- MKL crashes with SIGILL +# otherwise). On GPU this is not needed. +# * 32B is heavy: make sure you actually have the VRAM (bf16 -> ~65GB just +# for weights) before loading on CUDA, or run on CPU with the env vars +# above (slow, and considerably slower per token than the 1.5B model). +# ============================================================================== + +import os +import sys +import torch +from transformers import AutoTokenizer + +sys.path.append(os.getcwd()) +from JiRackTernaryUltra_1b import JiRackTransformer, JiRackConfig + +# ========================= EDIT THESE ========================= +MODEL_PATH = "/mnt/nfs_share/DeepSeek_1b/ds1p5b_checkpoint_migrated.pt" +TOKENIZER_DIR = "/mnt/nfs_share/DeepSeek_1b" +NO_THINK = True # True = skip reasoning # your extended tokenizer folder +# ================================================================ + + +def load_model(model_path: str): + device = "cuda" if torch.cuda.is_available() else "cpu" + print(f"🚀 Загрузка модели на устройство: {device.upper()}") + + config = JiRackConfig() + model = JiRackTransformer(config, use_checkpoint=False) + + print(f"📥 Загрузка весов из {model_path}...") + try: + ckpt = torch.load(model_path, map_location="cpu", weights_only=False) + state_dict = ckpt["model"] if isinstance(ckpt, dict) and "model" in ckpt else ckpt + + missing, unexpected = model.load_state_dict(state_dict, strict=False) + real_missing = [k for k in missing if not k.endswith("lambda_")] + if real_missing: + print(f"⚠️ Пропущено ключей: {len(real_missing)} -> {real_missing[:10]}") + if unexpected: + print(f"⚠️ Лишние ключи: {len(unexpected)} -> {unexpected[:10]}") + except Exception as e: + print(f"❌ Критическая ошибка при загрузке весов: {e}") + sys.exit(1) + + model = model.to(dtype=torch.bfloat16, device=device).eval() + model.set_lambda(0.0) # full-precision fast path, no fake-quant at inference + + if device == "cuda": + vram = torch.cuda.memory_allocated(0) / 1024**3 + print(f"✅ VRAM занято: {vram:.1f} GB") + else: + print("⚠️ ВНИМАНИЕ: Запуск 32B на CPU будет ОЧЕНЬ медленным.") + print(" Проверь, что выставлены MKL_ENABLE_INSTRUCTIONS=AVX и") + print(" MKL_DEBUG_CPU_TYPE=5 перед запуском (см. комментарий в шапке файла).") + + print("✅ Модель успешно загружена.") + return model, device + + +@torch.no_grad() +def generate_text(model, tokenizer, input_ids, stop_tokens, max_new_tokens=512, device="cuda"): + curr_ids = input_ids.to(device) + prompt_len = curr_ids.shape[1] + printed = "" + + temperature = 0.6 + top_p = 0.95 + repetition_penalty = 1.15 + + print("JiRack: ", end="", flush=True) + + for _ in range(max_new_tokens): + with torch.autocast(device_type=("cuda" if device == "cuda" else "cpu"), dtype=torch.bfloat16): + logits = model(curr_ids) + next_token_logits = logits[:, -1, :].float() / temperature + + for token_id in set(curr_ids[0].tolist()): + if next_token_logits[0, token_id] < 0: + next_token_logits[0, token_id] *= repetition_penalty + else: + next_token_logits[0, token_id] /= repetition_penalty + + sorted_logits, sorted_indices = torch.sort(next_token_logits, descending=True) + cumulative_probs = torch.cumsum(torch.softmax(sorted_logits, dim=-1), dim=-1) + sorted_indices_to_remove = cumulative_probs > top_p + sorted_indices_to_remove[..., 1:] = sorted_indices_to_remove[..., :-1].clone() + sorted_indices_to_remove[..., 0] = 0 + + next_token_logits[0, sorted_indices[sorted_indices_to_remove]] = -float('Inf') + probs = torch.softmax(next_token_logits, dim=-1) + next_token = torch.multinomial(probs, num_samples=1) + + curr_ids = torch.cat([curr_ids, next_token], dim=1) + # decode the whole generated tail each step and print only the new part; + # this keeps multi-token UTF-8 chars (emoji etc.) intact instead of \ufffd + decoded = tokenizer.decode(curr_ids[0, prompt_len:], skip_special_tokens=True) + if not decoded.endswith("\ufffd"): + print(decoded[len(printed):], end="", flush=True) + printed = decoded + + if next_token.item() in stop_tokens: + break + + print("\n") + return curr_ids + + +def main(): + if not os.path.exists(MODEL_PATH): + print(f"❌ Файл {MODEL_PATH} не найден!") + return + + try: + tokenizer = AutoTokenizer.from_pretrained(TOKENIZER_DIR) + except Exception as e: + print(f"❌ Ошибка токенайзера: {e}") + return + + model, device = load_model(MODEL_PATH) + + stop_tokens = set() + if tokenizer.eos_token_id is not None: + stop_tokens.add(tokenizer.eos_token_id) + for name in ("<|end_of_sentence|>", "<|endoftext|>", "<|im_end|>"): + tid = tokenizer.convert_tokens_to_ids(name) + if tid is not None and tid != tokenizer.unk_token_id: + stop_tokens.add(tid) + + print("\n" + "=" * 80) + print("✅ JiRack 32B (DeepSeek-R1-Distill-Qwen, extended tokenizer) Ready") + print("=" * 80 + "\n") + + history = [] + + while True: + try: + user_input = input("User: ") + if user_input.lower() in ["exit", "quit", "q"]: + break + if not user_input.strip(): + continue + + history.append({"role": "user", "content": user_input}) + + input_ids = tokenizer.apply_chat_template( + history, + add_generation_prompt=True, + return_tensors="pt", + return_dict=False, + ) + # some transformers versions return a BatchEncoding here regardless; + # unwrap it defensively so we always end up with a plain tensor + if not torch.is_tensor(input_ids): + input_ids = input_ids["input_ids"] + + if NO_THINK: + close_ids = tokenizer.encode("\n\n", add_special_tokens=False, return_tensors="pt") + input_ids = torch.cat([input_ids, close_ids], dim=1) + + curr_ids = generate_text(model, tokenizer, input_ids, stop_tokens, device=device) + + new_tokens = curr_ids[0, input_ids.shape[1]:] + reply = tokenizer.decode(new_tokens, skip_special_tokens=True) + history.append({"role": "assistant", "content": reply}) + + except KeyboardInterrupt: + print("\nStopped.") + break + except Exception as e: + print(f"\n❌ Ошибка: {e}") + + +if __name__ == "__main__": + main() diff --git a/chat_template.jinja b/chat_template.jinja new file mode 100644 index 0000000..c2066bd --- /dev/null +++ b/chat_template.jinja @@ -0,0 +1 @@ +{% if not add_generation_prompt is defined %}{% set add_generation_prompt = false %}{% endif %}{% set ns = namespace(is_first=false, is_tool=false, is_output_first=true, system_prompt='') %}{%- for message in messages %}{%- if message['role'] == 'system' %}{% set ns.system_prompt = message['content'] %}{%- endif %}{%- endfor %}{{bos_token}}{{ns.system_prompt}}{%- for message in messages %}{%- if message['role'] == 'user' %}{%- set ns.is_tool = false -%}{{'<|User|>' + message['content']}}{%- endif %}{%- if message['role'] == 'assistant' and message['content'] is none %}{%- set ns.is_tool = false -%}{%- for tool in message['tool_calls']%}{%- if not ns.is_first %}{{'<|Assistant|><|tool▁calls▁begin|><|tool▁call▁begin|>' + tool['type'] + '<|tool▁sep|>' + tool['function']['name'] + '\n' + '```json' + '\n' + tool['function']['arguments'] + '\n' + '```' + '<|tool▁call▁end|>'}}{%- set ns.is_first = true -%}{%- else %}{{'\n' + '<|tool▁call▁begin|>' + tool['type'] + '<|tool▁sep|>' + tool['function']['name'] + '\n' + '```json' + '\n' + tool['function']['arguments'] + '\n' + '```' + '<|tool▁call▁end|>'}}{{'<|tool▁calls▁end|><|end▁of▁sentence|>'}}{%- endif %}{%- endfor %}{%- endif %}{%- if message['role'] == 'assistant' and message['content'] is not none %}{%- if ns.is_tool %}{{'<|tool▁outputs▁end|>' + message['content'] + '<|end▁of▁sentence|>'}}{%- set ns.is_tool = false -%}{%- else %}{% set content = message['content'] %}{% if '' in content %}{% set content = content.split('')[-1] %}{% endif %}{{'<|Assistant|>' + content + '<|end▁of▁sentence|>'}}{%- endif %}{%- endif %}{%- if message['role'] == 'tool' %}{%- set ns.is_tool = true -%}{%- if ns.is_output_first %}{{'<|tool▁outputs▁begin|><|tool▁output▁begin|>' + message['content'] + '<|tool▁output▁end|>'}}{%- set ns.is_output_first = false %}{%- else %}{{'\n<|tool▁output▁begin|>' + message['content'] + '<|tool▁output▁end|>'}}{%- endif %}{%- endif %}{%- endfor -%}{% if ns.is_tool %}{{'<|tool▁outputs▁end|>'}}{% endif %}{% if add_generation_prompt and not ns.is_tool %}{{'<|Assistant|>\n'}}{% endif %} \ No newline at end of file diff --git a/config.json b/config.json new file mode 100644 index 0000000..cd6ccf1 --- /dev/null +++ b/config.json @@ -0,0 +1,21 @@ +{ + "architectures": [ + "Qwen2ForCausalLM" + ], + "model_type": "qwen2", + "vocab_size": 151936, + "hidden_size": 1536, + "intermediate_size": 8960, + "num_hidden_layers": 28, + "num_attention_heads": 12, + "num_key_value_heads": 2, + "hidden_act": "silu", + "max_position_embeddings": 131072, + "rms_norm_eps": 1e-06, + "rope_theta": 10000.0, + "tie_word_embeddings": false, + "torch_dtype": "bfloat16", + "use_cache": true, + "bos_token_id": 151646, + "eos_token_id": 151643 +} \ No newline at end of file diff --git a/generation_config.json b/generation_config.json new file mode 100644 index 0000000..47ddd75 --- /dev/null +++ b/generation_config.json @@ -0,0 +1,7 @@ +{ + "bos_token_id": 151646, + "eos_token_id": 151643, + "do_sample": true, + "temperature": 0.6, + "top_p": 0.95 +} \ No newline at end of file diff --git a/get_tool_call.py b/get_tool_call.py new file mode 100644 index 0000000..34774c6 --- /dev/null +++ b/get_tool_call.py @@ -0,0 +1,229 @@ +#!/usr/bin/env python3 +""" +Download ToolBench + APIGen-MT + ToolACE +and convert them to Qwen 2.5 SFT JSONL format +(with tool calling / function calling support). +""" + +import os +import json +import gzip +import tarfile +import zipfile +import requests +from pathlib import Path +from tqdm import tqdm +from datasets import load_dataset +from huggingface_hub import hf_hub_download, snapshot_download + +# ====================== CONFIG ====================== +OUTPUT_DIR = Path("./qwen25_tool_sft") +OUTPUT_DIR.mkdir(parents=True, exist_ok=True) + +FINAL_JSONL = OUTPUT_DIR / "tool_sft_qwen25.jsonl" + +# ==================================================== + +def download_file(url: str, dest: Path): + if dest.exists(): + print(f"[skip] {dest.name} already exists") + return + print(f"Downloading {url} ...") + with requests.get(url, stream=True) as r: + r.raise_for_status() + total = int(r.headers.get("content-length", 0)) + with open(dest, "wb") as f, tqdm(total=total, unit="B", unit_scale=True) as pbar: + for chunk in r.iter_content(chunk_size=8192): + f.write(chunk) + pbar.update(len(chunk)) + + +def to_qwen_messages(system: str | None, conversations: list[dict]) -> dict: + """ + Convert a list of turns into Qwen 2.5 messages format. + conversations: list of {"from": "human/gpt/function/...", "value": "..."} + """ + messages = [] + if system: + messages.append({"role": "system", "content": system}) + + for turn in conversations: + role = turn.get("from", "").lower() + content = turn.get("value", "").strip() + if not content: + continue + + if role in ("human", "user"): + messages.append({"role": "user", "content": content}) + elif role in ("gpt", "assistant"): + messages.append({"role": "assistant", "content": content}) + elif role in ("function", "tool", "observation"): + # Qwen-style tool response + messages.append({"role": "tool", "content": content}) + else: + # fallback + messages.append({"role": "user", "content": content}) + + return {"messages": messages} + + +# ---------------------------------------------------- +# 1. ToolBench (official) +# ---------------------------------------------------- +def process_toolbench(): + print("\n=== ToolBench ===") + # ToolBench is available on Hugging Face + try: + ds = load_dataset("ToolBench/ToolBench", split="train", trust_remote_code=True) + except Exception: + # fallback to the processed version that many people use + ds = load_dataset("lmsys/toolbench", split="train") + + count = 0 + with open(FINAL_JSONL, "a", encoding="utf-8") as fout: + for sample in tqdm(ds, desc="ToolBench"): + # ToolBench usually has "conversations" or "messages" + convs = sample.get("conversations") or sample.get("messages") or [] + if not convs: + continue + + # Some versions already have role/content + if isinstance(convs[0], dict) and "role" in convs[0]: + messages = [] + for m in convs: + role = m.get("role", "user") + content = m.get("content", "") + if role == "function": + role = "tool" + messages.append({"role": role, "content": content}) + record = {"messages": messages} + else: + record = to_qwen_messages(None, convs) + + if len(record["messages"]) >= 2: + fout.write(json.dumps(record, ensure_ascii=False) + "\n") + count += 1 + print(f"ToolBench → {count} samples") + + +# ---------------------------------------------------- +# 2. APIGen-MT (multi-turn tool calling) +# ---------------------------------------------------- +def process_apigen_mt(): + print("\n=== APIGen-MT ===") + # Common locations / names + possible = [ + "Salesforce/APIGen-MT", + "Salesforce/xLAM-APIGen", + "Salesforce/APIGen", + ] + ds = None + for name in possible: + try: + ds = load_dataset(name, split="train") + print(f"Loaded {name}") + break + except Exception: + continue + + if ds is None: + print("APIGen-MT not found on HF under common names. Skipping.") + return + + count = 0 + with open(FINAL_JSONL, "a", encoding="utf-8") as fout: + for sample in tqdm(ds, desc="APIGen-MT"): + # APIGen usually has "messages" already close to OpenAI format + messages = sample.get("messages") or sample.get("conversations") + if not messages: + continue + + # Normalize role names + normalized = [] + for m in messages: + role = m.get("role", "user").lower() + content = m.get("content", "") + if role == "function": + role = "tool" + normalized.append({"role": role, "content": content}) + + if len(normalized) >= 2: + fout.write(json.dumps({"messages": normalized}, ensure_ascii=False) + "\n") + count += 1 + print(f"APIGen-MT → {count} samples") + + +# ---------------------------------------------------- +# 3. ToolACE +# ---------------------------------------------------- +def process_toolace(): + print("\n=== ToolACE ===") + possible = [ + "Team-ACE/ToolACE", + "ToolACE/ToolACE", + "microsoft/ToolACE", + ] + ds = None + for name in possible: + try: + ds = load_dataset(name, split="train") + print(f"Loaded {name}") + break + except Exception: + continue + + if ds is None: + print("ToolACE not found under common names. Trying alternative...") + # Some people host processed versions + try: + ds = load_dataset("json", data_files="https://huggingface.co/datasets/Team-ACE/ToolACE/resolve/main/data/train.json") + except Exception: + print("Could not load ToolACE. Skipping.") + return + + count = 0 + with open(FINAL_JSONL, "a", encoding="utf-8") as fout: + for sample in tqdm(ds, desc="ToolACE"): + messages = sample.get("messages") or sample.get("conversations") or [] + if not messages: + continue + + normalized = [] + for m in messages: + if isinstance(m, dict): + role = m.get("role", m.get("from", "user")).lower() + content = m.get("content", m.get("value", "")) + else: + continue + if role in ("function", "observation"): + role = "tool" + elif role in ("human", "user"): + role = "user" + elif role in ("gpt", "assistant"): + role = "assistant" + normalized.append({"role": role, "content": content}) + + if len(normalized) >= 2: + fout.write(json.dumps({"messages": normalized}, ensure_ascii=False) + "\n") + count += 1 + print(f"ToolACE → {count} samples") + + +# ---------------------------------------------------- +# Main +# ---------------------------------------------------- +if __name__ == "__main__": + # Clear previous output if you want a fresh file + if FINAL_JSONL.exists(): + print(f"Removing old {FINAL_JSONL}") + FINAL_JSONL.unlink() + + process_toolbench() + process_apigen_mt() + process_toolace() + + # Final stats + total = sum(1 for _ in open(FINAL_JSONL, "r", encoding="utf-8")) + print(f"\n✅ Done! Total samples written → {FINAL_JSONL}") + print(f" Total lines: {total}") + print("\nYou can now use this JSONL for Qwen2.5 SFT (tool calling / function calling).") diff --git a/gguf.txt b/gguf.txt new file mode 100644 index 0000000..5e27841 --- /dev/null +++ b/gguf.txt @@ -0,0 +1,14 @@ +# 1. Конвертация в bf16 (как договорились) +python convert_hf_to_gguf.py /mnt/nfs_share/JiRackUlrta_1 \ + --outfile /mnt/nfs_share/JiRackUlrta_1/jirack_1p5b.gguf --outtype bf16 + +# 2. Сборка llama-quantize (один раз) +cd /mnt/nfs_share/llama.cpp +cmake -B build +cmake --build build -j + +# 3. Квантование в Q4_K_M +./build/bin/llama-quantize \ + /mnt/nfs_share/JiRackUlrta_1/jirack_1p5b.gguf \ + /mnt/nfs_share/JiRackUlrta_1/jirack_1p5b.Q4_K_M.gguf \ + Q4_K_M diff --git a/gguf_chat.sh b/gguf_chat.sh new file mode 100644 index 0000000..6121dbd --- /dev/null +++ b/gguf_chat.sh @@ -0,0 +1,7 @@ +export PATH="/mnt/nfs_share/llama.cpp/build/bin:$PATH" + +#llama-cli -m JiRackUltra_1b.gguf -p "You are helpfull aasistent" -n 256 +#llama-cli -m JiRackUltra_1b_Q4_K_M.gguf -p "You are helpfull aasistent" -n 256 +llama-cli -m JiRackUltra_1b_Q3_K_M.gguf -p "You are helpfull aasistent" -n 256 +#llama-cli -m JiRackUltra_1b_Q2_K.gguf -p "You are helpfull aasistent" -n 256 + diff --git a/jirack_to_gguf_1p5b.py b/jirack_to_gguf_1p5b.py new file mode 100644 index 0000000..6751edf --- /dev/null +++ b/jirack_to_gguf_1p5b.py @@ -0,0 +1,371 @@ +# ============================================================================== +# JiRack -> GGUF converter, 1.5B edition (stage 1: .pt -> HuggingFace folder) +# COPYRIGHT (c) 2026 Konstantin Vladimirovich Grabko. +# +# Verified against JiRackTernaryUltra_1b.py [DS1.5-1]: +# vocab_size 151936, hidden 1536, n_layers 28, n_heads 12, n_kv_heads 2, +# head_dim 128 (12*128=1536 -- the converter's hardcoded 128 is correct), +# rope_theta 10000.0 (same as 7B), rms_eps 1e-6, +# tie_word_embeddings = FALSE [DS1.5-2] -- lm_head ships separately, and +# this converter auto-detects that from the presence of lm_head.weight. +# +# Pipeline is two stages: +# +# Stage 1 (THIS SCRIPT, run in venv_ji): +# model.pt -> HF folder (model.safetensors + config.json + tokenizer) +# +# Stage 2 (llama.cpp, run once per model): +# python convert_hf_to_gguf.py \ +# --outfile jirack_1p5b.gguf --outtype bf16 +# ./build/bin/llama-quantize jirack_1p5b.gguf \ +# jirack_1p5b.Q4_K_M.gguf Q4_K_M +# +# Key points handled here: +# * config.json is derived from the ACTUAL tensor shapes in the checkpoint, +# so vocab (151936 vs 7B's 152064) and any Net2Net-expanded FFN width are +# picked up automatically -- no stock-config copying. +# * lambda_ buffers (ternary fake-quant training machinery) are dropped -- +# at inference you run set_lambda(0.0) anyway, so the stored weights ARE +# the full-precision weights; the exported model is a plain Qwen2 dense. +# * Keys: HF naming passes through; JiRack native naming +# (token_emb / blocks.N.* / ffn_w1-w3-w2) is remapped automatically. +# +# EDIT THE THREE PATHS BELOW. +# ============================================================================== + +import json +import os +import re +import shutil +import sys + +import torch + +# ========================= EDIT THESE ========================= +CKPT_PATH = "model.pt" +TOKENIZER_DIR = "." +OUTPUT_DIR = "." +# rope_theta cannot be inferred from tensor shapes -- set per base model: +# DeepSeek-R1-Distill-Qwen-1.5B -> 10000.0 (same as 7B) +# DeepSeek-R1-Distill-Qwen-14B -> 1000000.0 +# DeepSeek-R1-Distill-Qwen-32B -> 1000000.0 +ROPE_THETA = 10000.0 +MAX_POSITION = 131072 +RMS_NORM_EPS = 1e-6 + +# Q2_0 = 2-bit ternary {-1, 0, +1} quantization, one fp16 scale per group of +# weights -- the real encoding for BitNet-style ternary weights, once inference +# actually runs true ternary rather than bf16 dense. For now Q4_K_M remains +# the practical choice; the Q2_0 command is just printed ready for later. +EMIT_Q2_0_CMD = True +Q2_0_GROUP = 64 # 64 = mainline llama.cpp, no fork needed. +# ================================================================ + +# HF Qwen2 key patterns we expect to find (N = layer index) +HF_LAYER_KEYS = [ + "model.layers.{n}.self_attn.q_proj.weight", + "model.layers.{n}.self_attn.q_proj.bias", + "model.layers.{n}.self_attn.k_proj.weight", + "model.layers.{n}.self_attn.k_proj.bias", + "model.layers.{n}.self_attn.v_proj.weight", + "model.layers.{n}.self_attn.v_proj.bias", + "model.layers.{n}.self_attn.o_proj.weight", + "model.layers.{n}.mlp.gate_proj.weight", + "model.layers.{n}.mlp.up_proj.weight", + "model.layers.{n}.mlp.down_proj.weight", + "model.layers.{n}.input_layernorm.weight", + "model.layers.{n}.post_attention_layernorm.weight", +] +HF_TOP_KEYS = [ + "model.embed_tokens.weight", + "model.norm.weight", + "lm_head.weight", +] + + +def load_state_dict(path): + print(f"📥 Loading checkpoint: {path}") + ckpt = torch.load(path, map_location="cpu", weights_only=False) + sd = ckpt["model"] if isinstance(ckpt, dict) and "model" in ckpt else ckpt + if not isinstance(sd, dict): + sys.exit("❌ Checkpoint is not a state_dict and has no 'model' key.") + return sd + + +def drop_training_buffers(sd): + dropped = [k for k in sd if k.endswith("lambda_")] + for k in dropped: + del sd[k] + if dropped: + print(f"🧹 Dropped {len(dropped)} lambda_ buffers (ternary training machinery).") + return sd + + +def normalize_keys(sd): + """Pass HF-style keys through; try trivial prefix fixes; else abort with a listing.""" + keys = list(sd.keys()) + + # Case 1: already HF-style + if "model.embed_tokens.weight" in sd: + print("✅ Keys already use HF (Qwen2) naming -- no remap needed.") + return sd + + # Case 2: same names but without the leading 'model.' (e.g. 'embed_tokens.weight') + if "embed_tokens.weight" in sd: + print("🔁 Keys look HF-like without the 'model.' prefix -- adding it.") + out = {} + for k, v in sd.items(): + if k == "lm_head.weight": + out[k] = v + else: + out["model." + k] = v + if "model.embed_tokens.weight" in out: + return out + + # Case 3: JiRack native naming (token_emb / blocks.N.* / ffn_w1-w3-w2) + if "token_emb.weight" in sd and any(k.startswith("blocks.") for k in sd): + print("🔁 JiRack native naming detected -- remapping to HF (Qwen2) keys.") + hidden = sd["token_emb.weight"].shape[1] + block_map = { + "norm1.weight": "input_layernorm.weight", + "norm2.weight": "post_attention_layernorm.weight", + "q_proj.weight": "self_attn.q_proj.weight", + "q_proj.bias": "self_attn.q_proj.bias", + "k_proj.weight": "self_attn.k_proj.weight", + "k_proj.bias": "self_attn.k_proj.bias", + "v_proj.weight": "self_attn.v_proj.weight", + "v_proj.bias": "self_attn.v_proj.bias", + "out_proj.weight": "self_attn.o_proj.weight", + "ffn_w1.weight": "mlp.gate_proj.weight", # SwiGLU gate + "ffn_w3.weight": "mlp.up_proj.weight", # SwiGLU up + "ffn_w2.weight": "mlp.down_proj.weight", # SwiGLU down + } + out = {"model.embed_tokens.weight": sd["token_emb.weight"]} + leftovers = {} + blk_pat = re.compile(r"^blocks\.(\d+)\.(.+)$") + for k, v in sd.items(): + if k == "token_emb.weight": + continue + m = blk_pat.match(k) + if m: + idx, sub = m.group(1), m.group(2) + if sub == "out_proj.bias": + sys.exit("❌ out_proj has a bias -- Qwen2 arch has no o_proj " + "bias, this checkpoint isn't Qwen2-compatible as-is.") + if sub not in block_map: + sys.exit(f"❌ Unknown per-block key: {k} -- send this back.") + out[f"model.layers.{idx}.{block_map[sub]}"] = v + else: + leftovers[k] = v + # classify the remaining top-level keys by tensor shape + for k, v in leftovers.items(): + shp = tuple(v.shape) + if len(shp) == 1 and shp[0] == hidden: + print(f" final norm : {k} -> model.norm.weight") + out["model.norm.weight"] = v + elif len(shp) == 2 and shp[1] == hidden: + print(f" lm head : {k} -> lm_head.weight") + out["lm_head.weight"] = v + else: + sys.exit(f"❌ Unexplained top-level key: {k} {shp} -- send back.") + if "model.norm.weight" not in out: + sys.exit("❌ No final-norm tensor found (1-D, size=hidden). Send the " + "full key list (the tail beyond the first 80).") + print(f"✅ Remapped {len(out)} tensors to HF naming.") + return out + + # Case 4: unknown naming -- print everything and stop + print("❌ Unrecognized key naming scheme. Full key list (first 80):") + for k in keys[:80]: + print(" ", k, tuple(sd[k].shape) if hasattr(sd[k], "shape") else "") + print(f" ... total {len(keys)} keys") + sys.exit( + "\nSend this key list back and I'll add the exact JiRack->HF mapping " + "to normalize_keys()." + ) + + +def infer_config(sd): + """Derive Qwen2 config.json entirely from tensor shapes.""" + embed = sd["model.embed_tokens.weight"] + vocab_size, hidden_size = embed.shape + + layer_ids = set() + pat = re.compile(r"^model\.layers\.(\d+)\.") + for k in sd: + m = pat.match(k) + if m: + layer_ids.add(int(m.group(1))) + num_layers = max(layer_ids) + 1 + + q_w = sd["model.layers.0.self_attn.q_proj.weight"] # [n_heads*head_dim, hidden] + k_w = sd["model.layers.0.self_attn.k_proj.weight"] # [n_kv*head_dim, hidden] + gate = sd["model.layers.0.mlp.gate_proj.weight"] # [intermediate, hidden] + intermediate_size = gate.shape[0] + + # Qwen2 1.5B/7B/14B/32B all use head_dim=128 (1.5B: 12*128=1536) + head_dim = 128 + num_attention_heads = q_w.shape[0] // head_dim + num_key_value_heads = k_w.shape[0] // head_dim + + # sanity: every layer's FFN must have the same (expanded) width + widths = {sd[f"model.layers.{i}.mlp.gate_proj.weight"].shape[0] for i in layer_ids} + if len(widths) != 1: + sys.exit(f"❌ Inconsistent FFN widths across layers: {sorted(widths)}") + + tie = "lm_head.weight" not in sd + cfg = { + "architectures": ["Qwen2ForCausalLM"], + "model_type": "qwen2", + "vocab_size": vocab_size, + "hidden_size": hidden_size, + "intermediate_size": intermediate_size, + "num_hidden_layers": num_layers, + "num_attention_heads": num_attention_heads, + "num_key_value_heads": num_key_value_heads, + "hidden_act": "silu", + "max_position_embeddings": MAX_POSITION, + "rms_norm_eps": RMS_NORM_EPS, + "rope_theta": ROPE_THETA, + "tie_word_embeddings": tie, + "torch_dtype": "bfloat16", + "use_cache": True, + "bos_token_id": 151646, + "eos_token_id": 151643, + } + print("🧾 Inferred config from tensor shapes:") + for k in ("vocab_size", "hidden_size", "intermediate_size", "num_hidden_layers", + "num_attention_heads", "num_key_value_heads", "tie_word_embeddings"): + print(f" {k} = {cfg[k]}") + print(f" rope_theta = {ROPE_THETA} (from the EDIT block -- verify for this base model!)") + return cfg + + +def save_hf(sd, cfg): + os.makedirs(OUTPUT_DIR, exist_ok=True) + + device = "cuda" if torch.cuda.is_available() else "cpu" + print(f"🔄 Casting weights to bf16 on {device.upper()} ...") + for k in sd: + t = sd[k] + if torch.is_tensor(t) and t.is_floating_point(): + sd[k] = t.to(device=device, dtype=torch.bfloat16).cpu().contiguous() + + try: + from safetensors.torch import save_file + # single-file safetensors; llama.cpp's converter handles it fine + path = os.path.join(OUTPUT_DIR, "model.safetensors") + print(f"💾 Saving {path} ...") + save_file(sd, path, metadata={"format": "pt"}) + except ImportError: + # fallback: pytorch_model.bin, also accepted by convert_hf_to_gguf.py + path = os.path.join(OUTPUT_DIR, "pytorch_model.bin") + print(f"⚠️ safetensors not installed -- saving {path} instead (also works).") + torch.save(sd, path) + + with open(os.path.join(OUTPUT_DIR, "config.json"), "w") as f: + json.dump(cfg, f, indent=2) + with open(os.path.join(OUTPUT_DIR, "generation_config.json"), "w") as f: + json.dump({"bos_token_id": cfg["bos_token_id"], + "eos_token_id": cfg["eos_token_id"], + "do_sample": True, "temperature": 0.6, "top_p": 0.95}, f, indent=2) + + print(f"📎 Copying tokenizer from {TOKENIZER_DIR} ...") + same_dir = os.path.abspath(TOKENIZER_DIR) == os.path.abspath(OUTPUT_DIR) + if same_dir: + print(" TOKENIZER_DIR == OUTPUT_DIR -- tokenizer files are already in " + "place, skipping copy.") + copied = sum( + 1 for name in os.listdir(TOKENIZER_DIR) + if name.startswith(("tokenizer", "special_tokens", "added_tokens", + "vocab", "merges", "chat_template")) + ) + else: + copied = 0 + for name in os.listdir(TOKENIZER_DIR): + if name.startswith(("tokenizer", "special_tokens", "added_tokens", "vocab", "merges", "chat_template")): + shutil.copy2(os.path.join(TOKENIZER_DIR, name), os.path.join(OUTPUT_DIR, name)) + copied += 1 + if copied == 0: + sys.exit(f"❌ No tokenizer files found in {TOKENIZER_DIR}") + print(f" copied {copied} tokenizer files.") + + +def verify(cfg): + """Cross-check tokenizer length vs embedding rows.""" + try: + from transformers import AutoTokenizer + tok = AutoTokenizer.from_pretrained(OUTPUT_DIR) + n = len(tok) + rows = cfg["vocab_size"] + if n > rows: + sys.exit(f"❌ Tokenizer has {n} tokens but embedding matrix only {rows} rows -- " + f"resize the checkpoint before converting.") + print(f"✅ Tokenizer check: {n} tokens <= {rows} embedding rows " + f"({rows - n} spare rows).") + except Exception as e: + print(f"⚠️ Could not verify tokenizer ({e}) -- continuing anyway.") + + +def main(): + if not os.path.exists(CKPT_PATH): + sys.exit(f"❌ {CKPT_PATH} not found") + sd = load_state_dict(CKPT_PATH) + sd = drop_training_buffers(sd) + sd = normalize_keys(sd) + cfg = infer_config(sd) + save_hf(sd, cfg) + verify(cfg) + + print("\n" + "=" * 78) + print("✅ Stage 1 done. HF model at:", OUTPUT_DIR) + print("=" * 78) + + out_norm = OUTPUT_DIR.rstrip("/") + gguf_base = "jirack_1p5b" if out_norm in ("", ".") else out_norm + + q2_0_block = "" + if EMIT_Q2_0_CMD: + suffix = "Q2_0" if Q2_0_GROUP == 64 else f"Q2_0_g{Q2_0_GROUP}" + fork_note = ( + "group-64 is in mainline llama.cpp -- no fork needed, CPU/Metal ready." + if Q2_0_GROUP == 64 else + "group-128 needs a CUDA fork -- not needed on CPU-only." + ) + q2_0_block = """ +Ternary quantization (Q2_0, 2 bits/weight, {{-1,0,+1}} + fp16 group scale -- +this is the real encoding for BitNet-style ternary weights, once your model +actually runs true ternary at inference rather than bf16 dense): + {fork_note} + + ./build/bin/llama-quantize {gguf} {gguf_q2} {suffix} +""".format( + fork_note=fork_note, + gguf=gguf_base + ".gguf", + gguf_q2=gguf_base + f".{suffix}.gguf", + suffix=suffix, + ) + + print(""" +Stage 2 -- make the GGUF (one-time llama.cpp setup, then per model): + + git clone https://github.com/ggml-org/llama.cpp /mnt/nfs_share/llama.cpp + cd /mnt/nfs_share/llama.cpp + pip install -r requirements.txt + + python convert_hf_to_gguf.py {out} \\ + --outfile {gguf} --outtype bf16 + +Optional dense quantization (build llama.cpp first: cmake -B build && cmake --build build -j): + + ./build/bin/llama-quantize {gguf} {gguf_q} Q4_K_M +{q2_0_block}""".format( + out=OUTPUT_DIR, + gguf=gguf_base + ".gguf", + gguf_q=gguf_base + ".Q4_K_M.gguf", + q2_0_block=q2_0_block, + )) + + +if __name__ == "__main__": + main() diff --git a/model.pt b/model.pt new file mode 100644 index 0000000..7b6cd15 --- /dev/null +++ b/model.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1967c7a23020b5a895ba2469c6d3e1768eb127a8d59e29cb8671cf95345e248d +size 3554344056 diff --git a/model.safetensors b/model.safetensors new file mode 100644 index 0000000..7064e30 --- /dev/null +++ b/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:16bea0a35dba47a078ac1750e001512a02837da37c70c3c806244efbe9efefc8 +size 3554214752 diff --git a/quant.sh b/quant.sh new file mode 100644 index 0000000..5a7f53d --- /dev/null +++ b/quant.sh @@ -0,0 +1,5 @@ +export PATH="/mnt/nfs_share/llama.cpp/build/bin:$PATH" + +#llama-quantize JiRackUltra_1b.gguf JiRackUltra_1b_Q4_K_M.gguf Q4_K_M +#llama-quantize JiRackUltra_1b.gguf JiRackUltra_1b_Q3_K_M.gguf Q3_K_M +llama-quantize JiRackUltra_1b.gguf JiRackUltra_1b-Q2_K.gguf Q2_K diff --git a/tokenizer.json b/tokenizer.json new file mode 100644 index 0000000..339b305 --- /dev/null +++ b/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fbc3d20619bd1b3199ccae9f2bfbbcf2035533ccc5e7488e3c79bc53fbfede7e +size 11443518 diff --git a/tokenizer_config.json b/tokenizer_config.json new file mode 100644 index 0000000..f97ca84 --- /dev/null +++ b/tokenizer_config.json @@ -0,0 +1,131 @@ +{ + "add_prefix_space": null, + "backend": "tokenizers", + "bos_token": "<|begin▁of▁sentence|>", + "clean_up_tokenization_spaces": false, + "eos_token": "<|end▁of▁sentence|>", + "is_local": false, + "legacy": true, + "local_files_only": false, + "model_max_length": 16384, + "pad_token": "<|end▁of▁sentence|>", + "sp_model_kwargs": {}, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null, + "additional_special_tokens": [ + "<|image|>", + "<|video|>", + "<|sound|>", + "<|voice|>", + "<|listening|>", + "<|vision|>", + "<|mood_happy|>", + "<|mood_sad|>", + "<|mood_angry|>", + "<|mood_neutral|>", + "", + "", + "", + "<|action_start|>", + "<|action_end|>", + "<|trajectory_start|>", + "<|trajectory_end|>", + "<|joint_start|>", + "<|joint_end|>", + "<|sensor_start|>", + "<|sensor_end|>", + "<|command_start|>", + "<|command_end|>", + "<|state_start|>", + "<|state_end|>", + "<|pose|>", + "<|velocity|>", + "<|force|>", + "<|torque|>", + "<|gripper|>", + "<|navigation|>", + "<|obstacle|>", + "<|task_start|>", + "<|task_end|>", + "<|plan_start|>", + "<|plan_end|>", + "<|behavior_start|>", + "<|behavior_end|>", + "<|skill_start|>", + "<|skill_end|>", + "<|motor|>", + "<|servo|>", + "<|imu|>", + "<|lidar|>", + "<|camera|>", + "<|depth|>", + "<|waypoint|>", + "<|path|>", + "<|collision|>", + "<|grasp|>", + "<|release|>", + "<|homing|>", + "<|emergency_stop|>", + "<|calibration|>", + "<|manipulation|>", + "<|locomotion|>", + "<|feedback|>", + "<|control_loop|>", + "<|language|>", + "<|tool_call_start|>", + "<|tool_call_end|>", + "<|tool_result_start|>", + "<|tool_result_end|>", + "__SCIENCE__", + "__CODING__", + "__STOCK_EXCHANGE__", + "__MEDICINE__", + "__GOVERNMENT__", + "__NEWS__", + "__GENERAL__", + "__MATERIAL_SCIENCE__", + "__ELECTRONICS__", + "__MICROELECTRONICS__", + "__ENGINEERING__", + "__ROBOTICS__", + "__ENERGY__", + "__AUTOMOTIVE__", + "__AVIATION__", + "__MATH__", + "__PYTHON__", + "__C__", + "__CPP__", + "__C_SHARP__", + "__JAVA__", + "__JAVASCRIPT__", + "__TYPESCRIPT__", + "__RUST__", + "__GO__", + "__RUBY__", + "__PHP__", + "__SWIFT__", + "__KOTLIN__", + "__BASH__", + "__SQL__", + "__ASSEMBLY__", + "__PHILOSOPHY__", + "__LITERATURE__", + "__SOCIOLOGY__", + "__PSYCHOLOGY__", + "__POLITICAL_SCIENCE__", + "__CULTURAL_STUDIES__", + "__ETHNOGRAPHY__", + "__HUMAN_RIGHTS__", + "__COMPLIANCE__", + "__MILITARY__", + "__BANKING__", + "__OIL_INDUSTRY__", + "__LIGHT_INDUSTRY__", + "__NATURE__", + "__OCEAN__", + "__SPORT__", + "__CULINARY__", + "__TRAVEL__", + "__HOBBY__" + ] +} \ No newline at end of file