commit 447d46fa4ca5980f6d652584223161cf69735959 Author: ModelHub XC Date: Thu Sep 10 18:46:16 2026 +0800 初始化项目,由ModelHub XC社区提供模型 Model: devwoo/Kybalion-1B Source: Original Platform diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..52373fe --- /dev/null +++ b/.gitattributes @@ -0,0 +1,36 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text +tokenizer.json filter=lfs diff=lfs merge=lfs -text diff --git a/README.md b/README.md new file mode 100644 index 0000000..5e24345 --- /dev/null +++ b/README.md @@ -0,0 +1,173 @@ +--- +language: +- en +license: llama3.2 +base_model: meta-llama/Llama-3.2-1B +tags: +- llama +- continued-pretraining +- sft +- lora +- 1b +- math +- code +- education +- small-llm +datasets: +- HuggingFaceFW/fineweb-edu +- open-web-math/open-web-math +- bigcode/starcoderdata +- HuggingFaceTB/cosmopedia +- teknium/OpenHermes-2.5 +- meta-math/MetaMathQA +- sahil2801/CodeAlpaca-20k +--- + +# Kybalion-1B + +**Kybalion-1B** is a 1B-parameter language model built on top of [Llama 3.2 1B](https://huggingface.co/meta-llama/Llama-3.2-1B) through a full **Continued Pre-Training (CPT) → Supervised Fine-Tuning (SFT)** pipeline, trained entirely on Google Colab A100. + +> **Why "Kybalion"?** +> The model was originally developed under the internal codename *Prometheus-1B*, but was renamed to *Kybalion-1B* before public release to avoid confusion with an existing model of the same name on HuggingFace. *Kybalion* refers to the ancient hermetic text symbolizing hidden knowledge — fitting for a model focused on education, mathematics, science, and code. + +--- + +## 🏆 Key Highlights + +- **Beats Llama-3.2-1B-Instruct** on HellaSwag (63.8% vs 61.1%) and ties on WinoGrande (62.4%) +- **4.5× GSM8K improvement** over TinyLlama-1.1B (10.8% vs 2.4%) — math pretraining works +- Outperforms TinyLlama-1.1B on **all 6 benchmarks** +- Trained by a single undergraduate student on consumer cloud hardware + +--- + +## 🔬 Key Contributions + +- Demonstrates that domain-balanced continued pretraining on curated multi-domain data (education, math, code, science) yields consistent improvements across commonsense reasoning benchmarks in 1B-scale models +- Suggests that multi-step mathematical reasoning remains a fundamental bottleneck for 1B-scale models, even when combining math-focused pretraining (OpenWebMath) with instruction tuning (MetaMathQA) +- Provides a fully reproducible, compute-efficient training recipe (CPT → LoRA SFT) built and executed **by a single undergraduate student in under one week**, demonstrating that meaningful LLM research is achievable without institutional resources or large teams + +--- + +## 📊 Benchmark Results + +All scores measured with [lm-evaluation-harness](https://github.com/EleutherAI/lm-evaluation-harness) under **identical conditions** (same prompts, same few-shot settings, same hardware). + +| Benchmark | TinyLlama-1.1B | Llama-3.2-1B-Instruct | **Kybalion-1B** | +|-----------|:--------------:|:---------------------:|:---------------:| +| MMLU | 25.0% | 46.1% | **32.0%** | +| ARC-C | 37.2% | 41.5% | **37.6%** | +| GSM8K | 2.4% | 33.5% | **10.8%** | +| HellaSwag | 61.2% | 61.1% | **63.8%** 🏆 | +| WinoGrande | 61.8% | 62.4% | **62.4%** 🏆 | +| TruthfulQA | 37.4% | 43.3% | **40.0%** | + +> 🏆 = outperforms Llama-3.2-1B-Instruct +> All evaluations run with `lm_eval.simple_evaluate()`, bfloat16, batch_size=8, A100 GPU. + +--- + +## 🔧 Training Pipeline + +### Phase 1: Continued Pre-Training (CPT) + +Fine-tuned the base weights of `meta-llama/Llama-3.2-1B` on ~3.5B tokens of curated multi-domain data. + +| Domain | Dataset | Ratio | Purpose | +|--------|---------|-------|---------| +| Education | FineWeb-Edu (score ≥ 3.0) | 35% | General knowledge & reasoning | +| Mathematics | OpenWebMath | 20% | Mathematical reasoning | +| Code | StarCoderData (Python) | 15% | Code generation | +| Textbook | Cosmopedia web_samples_v2 | 15% | Structured knowledge | +| Science | Cosmopedia stanford | 10% | Scientific reasoning | +| Story | Cosmopedia stories | 5% | Language fluency | + +**Training config:** +- Hardware: Google Colab A100 80GB +- Optimizer: AdamW, LR = 2e-5, Cosine decay, Warmup = 1000 steps +- Precision: BF16 +- Effective batch size: 32 (4 × 8 grad accum) +- Sequence length: 2048 (packed) +- Framework: HuggingFace `transformers.Trainer` (no Unsloth) + +### Phase 2: Supervised Fine-Tuning (SFT) + +Applied LoRA adapters to teach instruction-following, then merged into base weights. + +| Dataset | Size | Purpose | +|---------|------|---------| +| OpenHermes 2.5 | 100K | General instruction following | +| MetaMathQA | 50K | Mathematical reasoning (GSM8K boost) | +| CodeAlpaca | 20K | Code generation | + +**SFT config:** +- Method: LoRA (r=64, α=128, dropout=0.05) +- Target modules: q/k/v/o/gate/up/down proj (all linear layers) +- LR = 1e-4, Epochs = 3, Cosine decay +- Merged with `PeftModel.merge_and_unload()` for standalone deployment + +--- + +## 💻 Usage + +```python +from transformers import AutoTokenizer, AutoModelForCausalLM +import torch + +tokenizer = AutoTokenizer.from_pretrained("devwoo/Kybalion-1B") +model = AutoModelForCausalLM.from_pretrained( + "devwoo/Kybalion-1B", + torch_dtype=torch.bfloat16, + device_map="auto", +) + +def chat(user_message, system="You are a helpful and knowledgeable AI assistant."): + prompt = ( + f"<|begin_of_text|><|start_header_id|>system<|end_header_id|>\n\n" + f"{system}<|eot_id|>" + f"<|start_header_id|>user<|end_header_id|>\n\n" + f"{user_message}<|eot_id|>" + f"<|start_header_id|>assistant<|end_header_id|>\n\n" + ) + inputs = tokenizer(prompt, return_tensors="pt").to(model.device) + with torch.no_grad(): + outputs = model.generate( + **inputs, + max_new_tokens=512, + temperature=0.7, + top_p=0.9, + do_sample=True, + eos_token_id=tokenizer.convert_tokens_to_ids("<|eot_id|>"), + ) + return tokenizer.decode(outputs[0][inputs["input_ids"].shape[1]:], skip_special_tokens=True) + +print(chat("Explain the Pythagorean theorem and give an example.")) +print(chat("Write a Python function to check if a number is prime.")) +``` + +--- + +## 📦 GGUF Version + +A quantized **GGUF q4_k_m** version is available at [devwoo/Kybalion-1B-GGUF](https://huggingface.co/devwoo/Kybalion-1B-GGUF) for CPU/mobile inference with [llama.cpp](https://github.com/ggerganov/llama.cpp) or [Ollama](https://ollama.com). + +```bash +# With llama.cpp +./llama-cli -m Kybalion-1B-q4_k_m.gguf -p "Explain quantum computing." -n 256 +``` + +--- + +## ⚠️ Limitations + +- 1B parameters — smaller than most production models; may struggle with complex multi-step reasoning +- Not RLHF-aligned; may occasionally produce unhelpful or inconsistent responses +- English-only training data +- GSM8K score (10.8%) reflects room for improvement in math reasoning compared to larger models + +--- + +## 📄 License + +This model is derived from `meta-llama/Llama-3.2-1B` and follows the [Llama 3.2 Community License](https://ai.meta.com/llama/license/). +Training datasets are used under their respective open licenses. diff --git a/config.json b/config.json new file mode 100644 index 0000000..234059b --- /dev/null +++ b/config.json @@ -0,0 +1,36 @@ +{ + "architectures": [ + "LlamaForCausalLM" + ], + "attention_bias": false, + "attention_dropout": 0.0, + "bos_token_id": 128000, + "dtype": "bfloat16", + "eos_token_id": 128001, + "head_dim": 64, + "hidden_act": "silu", + "hidden_size": 2048, + "initializer_range": 0.02, + "intermediate_size": 8192, + "max_position_embeddings": 131072, + "mlp_bias": false, + "model_type": "llama", + "num_attention_heads": 32, + "num_hidden_layers": 16, + "num_key_value_heads": 8, + "pad_token_id": null, + "pretraining_tp": 1, + "rms_norm_eps": 1e-05, + "rope_parameters": { + "factor": 32.0, + "high_freq_factor": 4.0, + "low_freq_factor": 1.0, + "original_max_position_embeddings": 8192, + "rope_theta": 500000.0, + "rope_type": "llama3" + }, + "tie_word_embeddings": true, + "transformers_version": "5.0.0", + "use_cache": false, + "vocab_size": 128256 +} diff --git a/generation_config.json b/generation_config.json new file mode 100644 index 0000000..0f6e971 --- /dev/null +++ b/generation_config.json @@ -0,0 +1,9 @@ +{ + "_from_model_config": true, + "bos_token_id": 128000, + "do_sample": true, + "eos_token_id": 128001, + "temperature": 0.6, + "top_p": 0.9, + "transformers_version": "5.0.0" +} diff --git a/model.safetensors b/model.safetensors new file mode 100644 index 0000000..92e77f5 --- /dev/null +++ b/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f738ffc2eb02cb022e4f75d1b6389fb03a14c80d389453a7c4e84902a4b4ed5c +size 2471645608 diff --git a/tokenizer.json b/tokenizer.json new file mode 100644 index 0000000..1c1d8d5 --- /dev/null +++ b/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6b9e4e7fb171f92fd137b777cc2714bf87d11576700a1dcd7a399e7bbe39537b +size 17209920 diff --git a/tokenizer_config.json b/tokenizer_config.json new file mode 100644 index 0000000..35f0a89 --- /dev/null +++ b/tokenizer_config.json @@ -0,0 +1,14 @@ +{ + "backend": "tokenizers", + "bos_token": "<|begin_of_text|>", + "clean_up_tokenization_spaces": true, + "eos_token": "<|end_of_text|>", + "is_local": true, + "model_input_names": [ + "input_ids", + "attention_mask" + ], + "model_max_length": 131072, + "pad_token": "<|end_of_text|>", + "tokenizer_class": "TokenizersBackend" +}