From 68ec45eaf770b3d2201d95063455b3f600b26340 Mon Sep 17 00:00:00 2001 From: ModelHub XC Date: Mon, 3 Aug 2026 14:31:16 +0800 Subject: [PATCH] =?UTF-8?q?=E5=88=9D=E5=A7=8B=E5=8C=96=E9=A1=B9=E7=9B=AE?= =?UTF-8?q?=EF=BC=8C=E7=94=B1ModelHub=20XC=E7=A4=BE=E5=8C=BA=E6=8F=90?= =?UTF-8?q?=E4=BE=9B=E6=A8=A1=E5=9E=8B?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Model: nkthebass/TinyBrainBot-303m-instruct Source: Original Platform --- .gitattributes | 36 +++++++++ README.md | 118 ++++++++++++++++++++++++++++ added_tokens.json | 11 +++ config.json | 23 ++++++ model.safetensors | 3 + tinybrainbot-303m-instruct-F16.gguf | 3 + tokenizer.model | 3 + tokenizer_config.json | 10 +++ 8 files changed, 207 insertions(+) create mode 100644 .gitattributes create mode 100644 README.md create mode 100644 added_tokens.json create mode 100644 config.json create mode 100644 model.safetensors create mode 100644 tinybrainbot-303m-instruct-F16.gguf create mode 100644 tokenizer.model create mode 100644 tokenizer_config.json diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..198bed5 --- /dev/null +++ b/.gitattributes @@ -0,0 +1,36 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text +tinybrainbot-303m-instruct-F16.gguf filter=lfs diff=lfs merge=lfs -text diff --git a/README.md b/README.md new file mode 100644 index 0000000..2dd28f2 --- /dev/null +++ b/README.md @@ -0,0 +1,118 @@ +--- +license: apache-2.0 +language: +- en +library_name: transformers +pipeline_tag: text-generation +tags: +- llama +- instruct +- chat +- conversational +- small-language-model +--- + +# TinyBrainBot 303M — Instruct + +A **303M-parameter chat assistant** built from scratch on a home server (2× NVIDIA Tesla P100): pretrained → fact-distilled → supervised fine-tuned. It's a tiny, honest little assistant — it answers common questions, follows simple instructions, holds a short conversation, and (often) **admits when it doesn't know** instead of bluffing. + +Base (pretrained-only) version: **TinyBrainBot 303M Base**. + +## Model details + +| | | +|---|---| +| Parameters | ~303M | +| Architecture | LLaMA-style (`LlamaForCausalLM`) — RoPE, RMSNorm, SwiGLU, GQA | +| Layers / hidden / heads | 24 / 1024 / 16 (4 KV heads) | +| FFN / vocab / context | 2816 / 32,000 / 1024 | +| Tied embeddings | Yes | +| Special tokens | `<\|user\|>` `<\|assistant\|>` `<\|system\|>` `<\|end\|>` | +| EOS token | `<\|end\|>` | + +## ⚠️ Chat template — use SPACES, not newlines + +This is the single most important detail. The tokenizer normalizes newlines to spaces, so the model was trained with **spaces** between turns. The bundled `chat_template` already does this — **use `apply_chat_template`** and it just works: + +```python +from transformers import AutoModelForCausalLM, AutoTokenizer +import torch + +tok = AutoTokenizer.from_pretrained("your-username/tinybrainbot-303m-instruct") +model = AutoModelForCausalLM.from_pretrained("your-username/tinybrainbot-303m-instruct", torch_dtype=torch.float16).eval() + +msgs = [{"role": "user", "content": "What is the capital of France?"}] +prompt = tok.apply_chat_template(msgs, tokenize=False, add_generation_prompt=True) +ids = tok(prompt, return_tensors="pt", add_special_tokens=False).input_ids +out = model.generate(ids, max_new_tokens=64, do_sample=True, + temperature=0.5, top_p=0.9, repetition_penalty=1.2, eos_token_id=7) +print(tok.decode(out[0][ids.shape[1]:], skip_special_tokens=True)) +# -> Paris. +``` + +The rendered format is (note the **spaces**, no `\n`): +``` +<|user|> {message} <|end|> <|assistant|> +``` +If you build prompts by hand or configure a UI (llama.cpp / Ollama / LM Studio / Jan), **make sure the separators are spaces and the stop token is `<|end|>`** — a template with literal `\n` will feed the model out-of-distribution tokens and it will ramble without stopping. + +## Recommended sampling + +Small models need a **lower temperature** than large ones (temp 1.0 makes this model incoherent). + +| Use case | temp | top-p | rep penalty | +|---|---|---|---| +| General chat (default) | 0.7 | 0.9 | 1.2 | +| More creative | 0.8 | 0.9 | 1.2 | +| Factual / reliable | 0.4–0.6 | 0.9 | 1.15 | + +## Training + +- **Base:** pretrained from scratch on FineWeb-Edu, Wikipedia, TinyStories, OpenWebText2, Orca-Math. +- **Fact distillation:** synthetic Q&A + short-fact datasets generated by a stronger teacher (Ling 2.6 Flash), mixed into continued pretraining to densify factual recall. +- **SFT:** ~33k instruction→response pairs plus short 2-turn conversations, distilled from the same teacher, spanning: general instructions, concise Q&A, explain/summarize/rewrite/list/compare/how-to, short creative writing, honest **"I don't know"** examples, and polite refusals. Loss masked to the assistant turns only. + +## Evaluation + +Standard multiple-choice benchmarks (log-likelihood scoring, n=200 subsample — **absolute numbers run a bit low vs. the official lm-eval-harness**; use them for relative comparison): + +| Benchmark | TinyBrainBot 303M | random | +|---|---|---| +| ARC-Easy (acc_norm) | ~44% | 25 | +| OpenBookQA (acc_norm) | ~31% | 25 | +| HellaSwag (acc_norm) | ~29% | 25 | +| WinoGrande (acc) | ~46% | 50 | +| ARC-Challenge (acc_norm) | ~24% | 25 | +| MMLU (acc) | ~23% | 25 | + +Roughly **GPT-2-small tier** on general reasoning, with a relative edge on knowledge QA (ARC / OpenBookQA) from the fact distillation. Near-random on MMLU, as expected at this scale. + +## Sampling + + (best of 5 samples, temperature tuned per prompt): + +Q: What is the largest planet in the solar system? +Jupiter. + +Q: Explain what photosynthesis is in one or two simple sentences. +Photosynthesis is how plants use sunlight, water, and air to make food (glucose) and release oxygen. It occurs in many types like green plants, algae, and trees. + +Q: Write a short poem about the moon. +Silver moon, slow and gold, hums low over the sleeping sky. It dreams in silver rivers, guarding dreams in morning light. + +Q: List three healthy breakfast ideas. +Greek yogurt with berries, whole wheat bread, cream cheese. + +Q: Give me one tip for staying focused while studying. +Use 25-minute focus blocks with 5-minute breaks, silence notifications, and a clear workspace to keep your mind fresh. + +## Limitations + +- **Fragile facts.** Sensitive to phrasing and capitalization; standard, well-formed questions work best. Confidently wrong on the long tail — pair with **retrieval (RAG)** for anything important. +- **Weak reasoning/math** — it's 303M. +- The "I don't know" and refusal behaviors are **helpful but not 100% reliable** (they were a small slice of SFT). +- English only. + +## License + +Apache-2.0 *(change if you prefer)*. diff --git a/added_tokens.json b/added_tokens.json new file mode 100644 index 0000000..d4ed9a7 --- /dev/null +++ b/added_tokens.json @@ -0,0 +1,11 @@ +{ + "<|user|>": 4, + "<|assistant|>": 5, + "<|system|>": 6, + "<|end|>": 7, + "<|mem_l1|>": 8, + "<|mem_l2|>": 9, + "<|mem_l3|>": 10, + "<|sep|>": 11, + "<|summary|>": 12 +} \ No newline at end of file diff --git a/config.json b/config.json new file mode 100644 index 0000000..1d4bad8 --- /dev/null +++ b/config.json @@ -0,0 +1,23 @@ +{ + "architectures": [ + "LlamaForCausalLM" + ], + "model_type": "llama", + "hidden_size": 1024, + "intermediate_size": 2816, + "num_hidden_layers": 24, + "num_attention_heads": 16, + "num_key_value_heads": 4, + "head_dim": 64, + "hidden_act": "silu", + "max_position_embeddings": 1024, + "rope_theta": 10000.0, + "rms_norm_eps": 1e-05, + "vocab_size": 32000, + "tie_word_embeddings": true, + "torch_dtype": "float16", + "bos_token_id": 0, + "eos_token_id": 7, + "pad_token_id": 2, + "unk_token_id": 3 +} \ No newline at end of file diff --git a/model.safetensors b/model.safetensors new file mode 100644 index 0000000..b59853c --- /dev/null +++ b/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:27ff48ecae9bfd72a7277de6f6a400001489fb9904457f50e86678c1418488dd +size 606726136 diff --git a/tinybrainbot-303m-instruct-F16.gguf b/tinybrainbot-303m-instruct-F16.gguf new file mode 100644 index 0000000..cd11848 --- /dev/null +++ b/tinybrainbot-303m-instruct-F16.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ce10903f30b4f89b3083f7afad4dc02fd7a22c9a496f5e6ae6c2cb95bf655367 +size 607583264 diff --git a/tokenizer.model b/tokenizer.model new file mode 100644 index 0000000..c5ce9e3 --- /dev/null +++ b/tokenizer.model @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ccd36462f4d7ee6f683ba11381ef285eb162fbf01b9ddf729436ee1a56dc35d4 +size 783766 diff --git a/tokenizer_config.json b/tokenizer_config.json new file mode 100644 index 0000000..ad7116a --- /dev/null +++ b/tokenizer_config.json @@ -0,0 +1,10 @@ +{ + "tokenizer_class": "LlamaTokenizer", + "bos_token": "", + "eos_token": "<|end|>", + "pad_token": "", + "unk_token": "", + "add_bos_token": false, + "add_eos_token": false, + "chat_template": "{%- for m in messages -%}{{- '<|' + m['role'] + '|> ' + m['content'] + ' <|end|> ' -}}{%- endfor -%}{%- if add_generation_prompt -%}{{- '<|assistant|>' -}}{%- endif -%}" +} \ No newline at end of file