From 4492e3508c8424eb327c1a65447e67dc945b895f Mon Sep 17 00:00:00 2001 From: ModelHub XC Date: Thu, 10 Sep 2026 07:32:16 +0800 Subject: [PATCH] =?UTF-8?q?=E5=88=9D=E5=A7=8B=E5=8C=96=E9=A1=B9=E7=9B=AE?= =?UTF-8?q?=EF=BC=8C=E7=94=B1ModelHub=20XC=E7=A4=BE=E5=8C=BA=E6=8F=90?= =?UTF-8?q?=E4=BE=9B=E6=A8=A1=E5=9E=8B?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Model: Dingdust/VibeThinker-3B-heretic Source: Original Platform --- .gitattributes | 43 +++ README.md | 257 +++++++++++++++ chat_template.jinja | 54 ++++ config.json | 69 ++++ generation_config.json | 6 + model-00001-of-00007.safetensors | 3 + model-00002-of-00007.safetensors | 3 + model-00003-of-00007.safetensors | 3 + model-00004-of-00007.safetensors | 3 + model-00005-of-00007.safetensors | 3 + model-00006-of-00007.safetensors | 3 + model-00007-of-00007.safetensors | 3 + model.safetensors.index.json | 442 ++++++++++++++++++++++++++ pictures/Abstrct.png | 3 + pictures/Acc_and_Scale.png | 3 + pictures/Architecture.png | 3 + pictures/LeetCode.png | 3 + pictures/VibeThiinker-3B.png | 3 + pictures/VibeThinker-3B+CLR.png | 3 + pictures/animation-VibeThinker-3B.gif | 3 + tokenizer.json | 3 + tokenizer_config.json | 16 + 22 files changed, 932 insertions(+) create mode 100644 .gitattributes create mode 100644 README.md create mode 100644 chat_template.jinja create mode 100644 config.json create mode 100644 generation_config.json create mode 100644 model-00001-of-00007.safetensors create mode 100644 model-00002-of-00007.safetensors create mode 100644 model-00003-of-00007.safetensors create mode 100644 model-00004-of-00007.safetensors create mode 100644 model-00005-of-00007.safetensors create mode 100644 model-00006-of-00007.safetensors create mode 100644 model-00007-of-00007.safetensors create mode 100644 model.safetensors.index.json create mode 100644 pictures/Abstrct.png create mode 100644 pictures/Acc_and_Scale.png create mode 100644 pictures/Architecture.png create mode 100644 pictures/LeetCode.png create mode 100644 pictures/VibeThiinker-3B.png create mode 100644 pictures/VibeThinker-3B+CLR.png create mode 100644 pictures/animation-VibeThinker-3B.gif create mode 100644 tokenizer.json create mode 100644 tokenizer_config.json diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..0aca2cd --- /dev/null +++ b/.gitattributes @@ -0,0 +1,43 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text +tokenizer.json filter=lfs diff=lfs merge=lfs -text +animation-VibeThinker-3B.gif filter=lfs diff=lfs merge=lfs -text +pictures/Abstrct.png filter=lfs diff=lfs merge=lfs -text +pictures/Acc_and_Scale.png filter=lfs diff=lfs merge=lfs -text +pictures/Architecture.png filter=lfs diff=lfs merge=lfs -text +pictures/LeetCode.png filter=lfs diff=lfs merge=lfs -text +pictures/VibeThiinker-3B.png filter=lfs diff=lfs merge=lfs -text +pictures/VibeThinker-3B+CLR.png filter=lfs diff=lfs merge=lfs -text diff --git a/README.md b/README.md new file mode 100644 index 0000000..ccb70e8 --- /dev/null +++ b/README.md @@ -0,0 +1,257 @@ +--- +license: mit +language: +- en +base_model: +- Qwen/Qwen2.5-Coder-3B +tags: +- math +- code +- reasoning +- gpqa +- instruction-following +- heretic +- uncensored +- decensored +- abliterated +pipeline_tag: text-generation +library_name: transformers +--- +# This is a decensored version of a model, made using [Heretic](https://heretic-project.org) v1.4.0 + +## Abliteration parameters + +| Parameter | Value | +| :-------- | :---: | +| **direction_index** | 23.30 | +| **attn.o_proj.max_weight** | 1.26 | +| **attn.o_proj.max_weight_position** | 21.83 | +| **attn.o_proj.min_weight** | 1.20 | +| **attn.o_proj.min_weight_distance** | 13.99 | +| **mlp.down_proj.max_weight** | 1.49 | +| **mlp.down_proj.max_weight_position** | 21.32 | +| **mlp.down_proj.min_weight** | 0.86 | +| **mlp.down_proj.min_weight_distance** | 13.56 | + +## Performance + +| Metric | This model | Original model (a model) | +| :----- | :--------: | :---------------------------: | +| **KL divergence** | 0.0255 | 0 *(by definition)* | +| **Refusals** | 9/100 | 64/100 | + +## Residual Geometry + +| Layer | S(g,b) | S(g*,b*) | S(g,r) | S(g*,r*) | S(b,r) | S(b*,r*) | \|g\| | \|g*\| | \|b\| | \|b*\| | \|r\| | \|r*\| | Silh | +| ----- | ------ | -------- | ------- | -------- | ------ | -------- | ------ | ------ | ------ | ------ | ----- | ------ | ------ | +| 1 | 0.9994 | 0.9994 | 0.1419 | 0.1613 | 0.1773 | 0.1966 | 12.07 | 12.06 | 12.14 | 12.14 | 0.44 | 0.44 | 0.0582 | +| 2 | 0.9994 | 0.9994 | 0.1141 | 0.1236 | 0.1484 | 0.1577 | 15.81 | 15.80 | 15.88 | 15.88 | 0.55 | 0.55 | 0.0670 | +| 3 | 0.9987 | 0.9987 | 0.0273 | 0.0306 | 0.0785 | 0.0816 | 16.28 | 16.28 | 16.33 | 16.33 | 0.84 | 0.83 | 0.0788 | +| 4 | 0.9979 | 0.9979 | 0.0460 | 0.0506 | 0.1110 | 0.1152 | 16.28 | 16.28 | 16.36 | 16.36 | 1.07 | 1.06 | 0.0868 | +| 5 | 0.9973 | 0.9974 | 0.1027 | 0.1029 | 0.1750 | 0.1743 | 16.83 | 16.83 | 17.00 | 17.01 | 1.25 | 1.23 | 0.0941 | +| 6 | 0.9975 | 0.9976 | 0.0628 | 0.0623 | 0.1332 | 0.1319 | 17.63 | 17.63 | 17.75 | 17.76 | 1.26 | 1.24 | 0.0956 | +| 7 | 0.9964 | 0.9964 | -0.0058 | -0.0099 | 0.0794 | 0.0748 | 18.16 | 18.18 | 18.22 | 18.23 | 1.55 | 1.54 | 0.0870 | +| 8 | 0.9941 | 0.9941 | 0.0083 | 0.0045 | 0.1172 | 0.1131 | 18.91 | 18.93 | 19.04 | 19.06 | 2.07 | 2.07 | 0.0952 | +| 9 | 0.9849 | 0.9852 | -0.0096 | -0.0132 | 0.1636 | 0.1582 | 19.40 | 19.44 | 19.66 | 19.68 | 3.40 | 3.37 | 0.1217 | +| 10 | 0.9854 | 0.9857 | -0.0042 | -0.0089 | 0.1660 | 0.1596 | 20.51 | 20.56 | 20.80 | 20.83 | 3.54 | 3.51 | 0.1232 | +| 11 | 0.9869 | 0.9871 | 0.0023 | -0.0012 | 0.1637 | 0.1587 | 20.92 | 20.97 | 21.20 | 21.23 | 3.42 | 3.39 | 0.1262 | +| 12 | 0.9860 | 0.9862 | -0.0215 | -0.0283 | 0.1455 | 0.1377 | 25.71 | 25.78 | 25.98 | 26.02 | 4.33 | 4.31 | 0.1191 | +| 13 | 0.9838 | 0.9839 | -0.1014 | -0.1188 | 0.0786 | 0.0605 | 25.86 | 25.96 | 25.80 | 25.82 | 4.65 | 4.65 | 0.1191 | +| 14 | 0.9837 | 0.9839 | -0.0898 | -0.1108 | 0.0906 | 0.0685 | 26.00 | 26.12 | 26.00 | 26.02 | 4.69 | 4.68 | 0.1131 | +| 15 | 0.9830 | 0.9833 | -0.0709 | -0.0856 | 0.1133 | 0.0971 | 26.07 | 26.18 | 26.17 | 26.21 | 4.81 | 4.79 | 0.1106 | +| 16 | 0.9808 | 0.9813 | 0.0049 | -0.0102 | 0.1995 | 0.1825 | 26.02 | 26.17 | 26.56 | 26.61 | 5.17 | 5.13 | 0.1142 | +| 17 | 0.9858 | 0.9862 | 0.0985 | 0.0796 | 0.2640 | 0.2438 | 29.96 | 30.13 | 30.91 | 30.97 | 5.21 | 5.15 | 0.1160 | +| 18 | 0.9857 | 0.9861 | 0.0173 | 0.0000 | 0.1854 | 0.1664 | 29.06 | 29.21 | 29.57 | 29.63 | 4.98 | 4.93 | 0.1078 | +| 19 | 0.9863 | 0.9867 | 0.0289 | 0.0055 | 0.1935 | 0.1679 | 31.16 | 31.35 | 31.75 | 31.80 | 5.24 | 5.17 | 0.1101 | +| 20 | 0.9856 | 0.9860 | 0.0036 | -0.0164 | 0.1727 | 0.1508 | 30.69 | 30.85 | 31.16 | 31.20 | 5.27 | 5.21 | 0.1071 | +| 21 | 0.9810 | 0.9813 | 0.1356 | 0.1211 | 0.3253 | 0.3100 | 36.41 | 36.57 | 38.15 | 38.19 | 7.47 | 7.41 | 0.1585 | +| 22 | 0.9789 | 0.9792 | 0.1084 | 0.0922 | 0.3092 | 0.2923 | 38.37 | 38.60 | 40.11 | 40.19 | 8.24 | 8.19 | 0.1662 | +| 23 | 0.9724 | 0.9727 | 0.0068 | -0.0035 | 0.2401 | 0.2288 | 37.52 | 37.71 | 38.65 | 38.74 | 9.02 | 8.99 | 0.1715 | +| 24 | 0.9685 | 0.9685 | -0.0140 | -0.0286 | 0.2355 | 0.2212 | 40.01 | 40.24 | 41.17 | 41.24 | 10.25 | 10.27 | 0.1857 | +| 25 | 0.9549 | 0.9544 | -0.1280 | -0.1385 | 0.1722 | 0.1636 | 42.07 | 42.28 | 42.35 | 42.44 | 12.68 | 12.80 | 0.2114 | +| 26 | 0.9512 | 0.9504 | -0.1326 | -0.1426 | 0.1798 | 0.1723 | 44.42 | 44.64 | 44.76 | 44.85 | 13.94 | 14.09 | 0.2190 | +| 27 | 0.9524 | 0.9513 | -0.1822 | -0.1915 | 0.1263 | 0.1202 | 51.97 | 52.19 | 51.51 | 51.60 | 15.97 | 16.20 | 0.2333 | +| 28 | 0.9331 | 0.9309 | -0.1472 | -0.1589 | 0.2182 | 0.2126 | 53.06 | 53.29 | 53.77 | 53.85 | 19.55 | 19.92 | 0.2326 | +| 29 | 0.9280 | 0.9251 | -0.1168 | -0.1313 | 0.2617 | 0.2551 | 61.41 | 61.77 | 63.19 | 63.33 | 23.71 | 24.26 | 0.2238 | +| 30 | 0.9122 | 0.9090 | -0.1673 | -0.1811 | 0.2514 | 0.2452 | 66.16 | 66.62 | 67.40 | 67.58 | 28.01 | 28.64 | 0.2210 | +| 31 | 0.9066 | 0.9034 | -0.1648 | -0.1825 | 0.2669 | 0.2567 | 82.24 | 83.01 | 84.17 | 84.45 | 36.02 | 36.82 | 0.2197 | +| 32 | 0.9124 | 0.9102 | -0.0878 | -0.1084 | 0.3275 | 0.3132 | 112.01 | 113.23 | 118.09 | 118.53 | 48.51 | 49.39 | 0.2145 | +| 33 | 0.9271 | 0.9252 | -0.1205 | -0.1459 | 0.2604 | 0.2405 | 146.74 | 148.56 | 150.88 | 151.41 | 56.96 | 58.08 | 0.2068 | +| 34 | 0.9447 | 0.9435 | -0.0870 | -0.1083 | 0.2446 | 0.2272 | 177.54 | 179.19 | 182.41 | 182.92 | 60.06 | 60.97 | 0.2024 | +| 35 | 0.9363 | 0.9350 | -0.1146 | -0.1319 | 0.2415 | 0.2282 | 180.56 | 182.04 | 184.84 | 185.34 | 65.32 | 66.30 | 0.2059 | +| 36 | 0.8726 | 0.8711 | -0.1135 | -0.1285 | 0.3862 | 0.3750 | 86.94 | 87.92 | 93.65 | 94.06 | 46.04 | 46.57 | 0.1981 | + +g = mean of residual vectors for good prompts + +g* = geometric median of residual vectors for good prompts + +b = mean of residual vectors for bad prompts + +b* = geometric median of residual vectors for bad prompts + +r = refusal direction for means (i.e., b - g) + +r* = refusal direction for geometric medians (i.e., b* - g*) + +S(x,y) = cosine similarity of x and y + +|x| = L2 norm of x + +Silh = Mean silhouette coefficient of residuals for good/bad clusters + +## Residual Vectors Visualization + +![VibeThinker-3B-animation](pictures/animation-VibeThinker-3B.gif) + +----- + + +# VibeThinker-3B + +

GitHub  |  Hugging Face  |  Technical Report

+ +## Introduction + +VibeThinker-3B is a further exploration of the VibeThinker series at the 3B-parameter scale, focusing on challenging reasoning tasks with clear verification signals, such as mathematics, coding, and STEM. By systematically optimizing the Spectrum-to-Signal Principle (SSP) post-training pipeline introduced in VibeThinker-1.5B, VibeThinker-3B achieves strong performance on AIME, HMMT, IMO-AnswerBench, LiveCodeBench, and recent LeetCode contests, reaching the performance range of top-tier frontier reasoning models, including Qwen3.6 Plus, Gemini 3 Pro, GLM-5, and Kimi K2.5, on verifiable reasoning benchmarks. + +Motivated by these observations, we propose the Parametric Compression-Coverage Hypothesis: different capabilities depend on model parameters in fundamentally different ways. Verifiable reasoning is closer to a highly compressible, parameter-dense capability, centered on multi-step reasoning, constraint satisfaction, self-correction, and answer verification. When the task space is sufficiently structured and feedback signals are sufficiently reliable, compact models may also carry near-frontier reasoning capabilities. In contrast, open-domain knowledge, general-purpose dialogue, and long-tail scenario understanding rely more heavily on large-scale parameters to broadly cover facts, concepts, and world knowledge. + +From VibeThinker-1.5B to VibeThinker-3B, our goal is not to build a small model that replaces large-scale models, but to examine the real boundaries of small models along specific capability dimensions. With VibeThinker-3B, we aim to show that small models should not be viewed merely as a compromise for reducing deployment costs. For capability domains with clear feedback and verification mechanisms, SLMs emerge as a promising research trajectory toward frontier-level performance that is fundamentally complementary to the traditional parameter scaling paradigm. + +![alt text](pictures/Abstrct.png) + +## Key Performance Data + +📏 In terms of reasoning accuracy relative to model scale, VibeThinker-3B reaches 76.4 on IMO-AnswerBench, a highly challenging benchmark with 400 IMO-level problems, with only 3B parameters, and improves to 80.6 with Claim-Level Reliability Assessment (CLR), a test-time scaling strategy for answer-verifiable reasoning tasks. This demonstrates that a model within a strictly small-model regime can reach the performance range of substantially larger models, such as DeepSeek V3.2 (78.3, 671B), GLM-5 (82.5, 744B), and Kimi K2.5 (81.8, 1T). + +![alt text](pictures/Acc_and_Scale.png) + +💡 VibeThinker-3B achieves strong results across mathematics, coding, knowledge, and instruction-following benchmarks. + +![alt text](pictures/VibeThiinker-3B.png) + +🔁 VibeThinker-3B achieves competitive results against first-tier reasoning models and reaches the performance range of top-tier systems on several verifiable reasoning benchmarks. + +![alt text](pictures/VibeThinker-3B+CLR.png) + +🏆 To further test the model's out-of-distribution performance, we evaluate VibeThinker-3B on recent unseen LeetCode weekly and biweekly contests (Python) from Apr. 25 to May 31, 2026. VibeThinker-3B passes **123/128** first-attempt submissions, corresponding to a **96.1%** acceptance rate. + +![alt text](pictures/LeetCode.png) + + +## Training Pipeline + +VibeThinker-3B follows the **Spectrum-to-Signal Principle (SSP)** introduced in VibeThinker-1.5B. The SFT stage constructs a broad spectrum of valid reasoning trajectories, while the RL stage amplifies correct reasoning signals using verifiable rewards. + +![alt text](pictures/Architecture.png) + +The training pipeline contains the following stages: + +1. **Curriculum-based two-stage SFT** + - Stage 1 focuses on broad capability coverage across math, code, STEM reasoning, general dialogue, and instruction following. + - Stage 2 shifts toward harder and longer-horizon reasoning samples. + - Diversity-Exploring Distillation is used to preserve multiple valid solution paths. + +2. **Multi-domain Reasoning RL** + - VibeThinker-3B reuses MaxEnt-Guided Policy Optimization (MGPO). + - RL is applied sequentially to math, code, and STEM reasoning tasks. + - Training uses a single 64K long-context window to preserve complete long-horizon reasoning trajectories. + +3. **Offline Self-Distillation** + - High-quality trajectories from Math, Code, and STEM RL checkpoints are filtered and distilled back into a unified student model. + - A learning-potential score is used to prioritize traces that are correct but not yet well modeled by the student. + +4. **Instruct RL** + - The final stage improves controllability on user-facing prompts. + - Rule-based validators and rubric-based reward models are used for format-sensitive and open-ended instruction data. + +## Usage Guidelines + +We recommend using VibeThinker-3B for competitive-style math, coding, STEM reasoning, and other tasks where the target answer can be verified. For broad open-domain knowledge tasks, larger general-purpose models may still be more suitable. + +For benchmark-style evaluation, the technical report uses vLLM with: + +- `temperature=1.0` +- `top_p=0.95` +- `top_k=-1` + +## Quick Start + +Required: **transformers>=4.54.0** + +Recommended for better inference performance: **vLLM==0.10.1 or SGLang>=0.4.9.post6** + +```python +from transformers import AutoModelForCausalLM, AutoTokenizer, GenerationConfig + + +class VibeThinker: + def __init__(self, model_path): + self.model_path = model_path + self.model = AutoModelForCausalLM.from_pretrained( + self.model_path, + low_cpu_mem_usage=True, + torch_dtype="bfloat16", + device_map="auto", + ) + self.tokenizer = AutoTokenizer.from_pretrained( + self.model_path, + trust_remote_code=True, + ) + + def infer_text(self, prompt): + messages = [{"role": "user", "content": prompt}] + text = self.tokenizer.apply_chat_template( + messages, + tokenize=False, + add_generation_prompt=True, + ) + model_inputs = self.tokenizer([text], return_tensors="pt").to(self.model.device) + + generation_config = dict( + max_new_tokens=102400, + do_sample=True, + temperature=1.0, + top_p=0.95, + top_k=None, + ) + generated_ids = self.model.generate( + **model_inputs, + generation_config=GenerationConfig(**generation_config), + ) + generated_ids = [ + output_ids[len(input_ids):] + for input_ids, output_ids in zip(model_inputs.input_ids, generated_ids) + ] + + return self.tokenizer.batch_decode( + generated_ids, + skip_special_tokens=True, + )[0] + + +if __name__ == "__main__": + model = VibeThinker("WeiboAI/VibeThinker-3B") + prompt = "Your Prompt" + print(model.infer_text(prompt)) +``` + +## License + +The model repository is licensed under the MIT License. + +## Citations & References + +If you use VibeThinker-3B in your research or product, please cite: + +```bibtex +@misc{xu2026vibethinker3bexploringfrontierverifiable, + title={VibeThinker-3B: Exploring the Frontier of Verifiable Reasoning in Small Language Models}, + author={Sen Xu and Shixi Liu and Wei Wang and Jixin Min and Yingwei Dai and Zhibin Yin and Yirong Chen and Xin Zhou and Junlin Zhang}, + year={2026}, + eprint={2606.16140}, + archivePrefix={arXiv}, + primaryClass={cs.AI}, + url={https://arxiv.org/abs/2606.16140}, +} +``` diff --git a/chat_template.jinja b/chat_template.jinja new file mode 100644 index 0000000..28028c0 --- /dev/null +++ b/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/config.json b/config.json new file mode 100644 index 0000000..19f1fd0 --- /dev/null +++ b/config.json @@ -0,0 +1,69 @@ +{ + "architectures": [ + "Qwen2ForCausalLM" + ], + "attention_dropout": 0.0, + "bos_token_id": 151643, + "dtype": "bfloat16", + "eos_token_id": 151643, + "hidden_act": "silu", + "hidden_size": 2048, + "initializer_range": 0.02, + "intermediate_size": 11008, + "layer_types": [ + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention" + ], + "max_position_embeddings": 131072, + "max_window_layers": 36, + "model_type": "qwen2", + "num_attention_heads": 16, + "num_hidden_layers": 36, + "num_key_value_heads": 2, + "pad_token_id": null, + "rms_norm_eps": 1e-06, + "rope_parameters": { + "rope_theta": 1000000.0, + "rope_type": "default" + }, + "sliding_window": null, + "tie_word_embeddings": true, + "transformers_version": "5.14.1", + "use_cache": false, + "use_sliding_window": false, + "vocab_size": 151936 +} diff --git a/generation_config.json b/generation_config.json new file mode 100644 index 0000000..b380b58 --- /dev/null +++ b/generation_config.json @@ -0,0 +1,6 @@ +{ + "bos_token_id": 151643, + "eos_token_id": 151643, + "max_new_tokens": 2048, + "transformers_version": "5.14.1" +} diff --git a/model-00001-of-00007.safetensors b/model-00001-of-00007.safetensors new file mode 100644 index 0000000..0bc4d73 --- /dev/null +++ b/model-00001-of-00007.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3b6cb803b8954f9b2b7efd057b83e575ee287e380d97a3d93a0ba5aae7a15423 +size 975733736 diff --git a/model-00002-of-00007.safetensors b/model-00002-of-00007.safetensors new file mode 100644 index 0000000..166d6b2 --- /dev/null +++ b/model-00002-of-00007.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:508605e4e1fe696e580456a635446162e37624b1682a61d1adb27f5a30db2258 +size 970020872 diff --git a/model-00003-of-00007.safetensors b/model-00003-of-00007.safetensors new file mode 100644 index 0000000..d0b5458 --- /dev/null +++ b/model-00003-of-00007.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:10e7602799b62c47116cdd934675026da87b8dbacd6e63e0dbee6ca96dc173de +size 988909616 diff --git a/model-00004-of-00007.safetensors b/model-00004-of-00007.safetensors new file mode 100644 index 0000000..084879f --- /dev/null +++ b/model-00004-of-00007.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:635b205643cd9b584bff98cac9d230c8213158e536cf5e101b63c300cdec7809 +size 970020952 diff --git a/model-00005-of-00007.safetensors b/model-00005-of-00007.safetensors new file mode 100644 index 0000000..dab2687 --- /dev/null +++ b/model-00005-of-00007.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:32df3f2004fbf31597c43980e61b26bf0588395452d44f4a38c19d94515b50bc +size 970020944 diff --git a/model-00006-of-00007.safetensors b/model-00006-of-00007.safetensors new file mode 100644 index 0000000..a2d9de7 --- /dev/null +++ b/model-00006-of-00007.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d60b60b83378808607b68f679f3bb37c570adda5f67cb2ec2e6a49cb7c7e92b5 +size 988909632 diff --git a/model-00007-of-00007.safetensors b/model-00007-of-00007.safetensors new file mode 100644 index 0000000..363bdc5 --- /dev/null +++ b/model-00007-of-00007.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1a7def49b1496bfa4eda358b1b0b219ea626ab60cedb05de4f995a07bee29100 +size 308310688 diff --git a/model.safetensors.index.json b/model.safetensors.index.json new file mode 100644 index 0000000..7ec45a9 --- /dev/null +++ b/model.safetensors.index.json @@ -0,0 +1,442 @@ +{ + "metadata": { + "total_parameters": 3085938688, + "total_size": 6171877376 + }, + "weight_map": { + "model.embed_tokens.weight": "model-00001-of-00007.safetensors", + "model.layers.0.input_layernorm.weight": "model-00001-of-00007.safetensors", + "model.layers.0.mlp.down_proj.weight": "model-00001-of-00007.safetensors", + "model.layers.0.mlp.gate_proj.weight": "model-00001-of-00007.safetensors", + "model.layers.0.mlp.up_proj.weight": "model-00001-of-00007.safetensors", + "model.layers.0.post_attention_layernorm.weight": "model-00001-of-00007.safetensors", + "model.layers.0.self_attn.k_proj.bias": "model-00001-of-00007.safetensors", + "model.layers.0.self_attn.k_proj.weight": "model-00001-of-00007.safetensors", + "model.layers.0.self_attn.o_proj.weight": "model-00001-of-00007.safetensors", + "model.layers.0.self_attn.q_proj.bias": "model-00001-of-00007.safetensors", + "model.layers.0.self_attn.q_proj.weight": "model-00001-of-00007.safetensors", + "model.layers.0.self_attn.v_proj.bias": "model-00001-of-00007.safetensors", + "model.layers.0.self_attn.v_proj.weight": "model-00001-of-00007.safetensors", + "model.layers.1.input_layernorm.weight": "model-00001-of-00007.safetensors", + "model.layers.1.mlp.down_proj.weight": "model-00001-of-00007.safetensors", + "model.layers.1.mlp.gate_proj.weight": "model-00001-of-00007.safetensors", + "model.layers.1.mlp.up_proj.weight": "model-00001-of-00007.safetensors", + "model.layers.1.post_attention_layernorm.weight": "model-00001-of-00007.safetensors", + "model.layers.1.self_attn.k_proj.bias": "model-00001-of-00007.safetensors", + "model.layers.1.self_attn.k_proj.weight": "model-00001-of-00007.safetensors", + "model.layers.1.self_attn.o_proj.weight": "model-00001-of-00007.safetensors", + "model.layers.1.self_attn.q_proj.bias": "model-00001-of-00007.safetensors", + "model.layers.1.self_attn.q_proj.weight": "model-00001-of-00007.safetensors", + "model.layers.1.self_attn.v_proj.bias": "model-00001-of-00007.safetensors", + "model.layers.1.self_attn.v_proj.weight": "model-00001-of-00007.safetensors", + "model.layers.10.input_layernorm.weight": "model-00003-of-00007.safetensors", + "model.layers.10.mlp.down_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.10.mlp.gate_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.10.mlp.up_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.10.post_attention_layernorm.weight": "model-00003-of-00007.safetensors", + "model.layers.10.self_attn.k_proj.bias": "model-00003-of-00007.safetensors", + "model.layers.10.self_attn.k_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.10.self_attn.o_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.10.self_attn.q_proj.bias": "model-00003-of-00007.safetensors", + "model.layers.10.self_attn.q_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.10.self_attn.v_proj.bias": "model-00003-of-00007.safetensors", + "model.layers.10.self_attn.v_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.11.input_layernorm.weight": "model-00003-of-00007.safetensors", + "model.layers.11.mlp.down_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.11.mlp.gate_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.11.mlp.up_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.11.post_attention_layernorm.weight": "model-00003-of-00007.safetensors", + "model.layers.11.self_attn.k_proj.bias": "model-00003-of-00007.safetensors", + "model.layers.11.self_attn.k_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.11.self_attn.o_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.11.self_attn.q_proj.bias": "model-00003-of-00007.safetensors", + "model.layers.11.self_attn.q_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.11.self_attn.v_proj.bias": "model-00003-of-00007.safetensors", + "model.layers.11.self_attn.v_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.12.input_layernorm.weight": "model-00003-of-00007.safetensors", + "model.layers.12.mlp.down_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.12.mlp.gate_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.12.mlp.up_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.12.post_attention_layernorm.weight": "model-00003-of-00007.safetensors", + "model.layers.12.self_attn.k_proj.bias": "model-00003-of-00007.safetensors", + "model.layers.12.self_attn.k_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.12.self_attn.o_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.12.self_attn.q_proj.bias": "model-00003-of-00007.safetensors", + "model.layers.12.self_attn.q_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.12.self_attn.v_proj.bias": "model-00003-of-00007.safetensors", + "model.layers.12.self_attn.v_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.13.input_layernorm.weight": "model-00003-of-00007.safetensors", + "model.layers.13.mlp.down_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.13.mlp.gate_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.13.mlp.up_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.13.post_attention_layernorm.weight": "model-00003-of-00007.safetensors", + "model.layers.13.self_attn.k_proj.bias": "model-00003-of-00007.safetensors", + "model.layers.13.self_attn.k_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.13.self_attn.o_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.13.self_attn.q_proj.bias": "model-00003-of-00007.safetensors", + "model.layers.13.self_attn.q_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.13.self_attn.v_proj.bias": "model-00003-of-00007.safetensors", + "model.layers.13.self_attn.v_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.14.input_layernorm.weight": "model-00003-of-00007.safetensors", + "model.layers.14.mlp.down_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.14.mlp.gate_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.14.mlp.up_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.14.post_attention_layernorm.weight": "model-00003-of-00007.safetensors", + "model.layers.14.self_attn.k_proj.bias": "model-00003-of-00007.safetensors", + "model.layers.14.self_attn.k_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.14.self_attn.o_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.14.self_attn.q_proj.bias": "model-00003-of-00007.safetensors", + "model.layers.14.self_attn.q_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.14.self_attn.v_proj.bias": "model-00003-of-00007.safetensors", + "model.layers.14.self_attn.v_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.15.input_layernorm.weight": "model-00003-of-00007.safetensors", + "model.layers.15.mlp.down_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.15.mlp.gate_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.15.mlp.up_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.15.post_attention_layernorm.weight": "model-00004-of-00007.safetensors", + "model.layers.15.self_attn.k_proj.bias": "model-00004-of-00007.safetensors", + "model.layers.15.self_attn.k_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.15.self_attn.o_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.15.self_attn.q_proj.bias": "model-00004-of-00007.safetensors", + "model.layers.15.self_attn.q_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.15.self_attn.v_proj.bias": "model-00004-of-00007.safetensors", + "model.layers.15.self_attn.v_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.16.input_layernorm.weight": "model-00004-of-00007.safetensors", + "model.layers.16.mlp.down_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.16.mlp.gate_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.16.mlp.up_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.16.post_attention_layernorm.weight": "model-00004-of-00007.safetensors", + "model.layers.16.self_attn.k_proj.bias": "model-00004-of-00007.safetensors", + "model.layers.16.self_attn.k_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.16.self_attn.o_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.16.self_attn.q_proj.bias": "model-00004-of-00007.safetensors", + "model.layers.16.self_attn.q_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.16.self_attn.v_proj.bias": "model-00004-of-00007.safetensors", + "model.layers.16.self_attn.v_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.17.input_layernorm.weight": "model-00004-of-00007.safetensors", + "model.layers.17.mlp.down_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.17.mlp.gate_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.17.mlp.up_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.17.post_attention_layernorm.weight": "model-00004-of-00007.safetensors", + "model.layers.17.self_attn.k_proj.bias": "model-00004-of-00007.safetensors", + "model.layers.17.self_attn.k_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.17.self_attn.o_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.17.self_attn.q_proj.bias": "model-00004-of-00007.safetensors", + "model.layers.17.self_attn.q_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.17.self_attn.v_proj.bias": "model-00004-of-00007.safetensors", + "model.layers.17.self_attn.v_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.18.input_layernorm.weight": "model-00004-of-00007.safetensors", + "model.layers.18.mlp.down_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.18.mlp.gate_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.18.mlp.up_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.18.post_attention_layernorm.weight": "model-00004-of-00007.safetensors", + "model.layers.18.self_attn.k_proj.bias": "model-00004-of-00007.safetensors", + "model.layers.18.self_attn.k_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.18.self_attn.o_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.18.self_attn.q_proj.bias": "model-00004-of-00007.safetensors", + "model.layers.18.self_attn.q_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.18.self_attn.v_proj.bias": "model-00004-of-00007.safetensors", + "model.layers.18.self_attn.v_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.19.input_layernorm.weight": "model-00004-of-00007.safetensors", + "model.layers.19.mlp.down_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.19.mlp.gate_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.19.mlp.up_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.19.post_attention_layernorm.weight": "model-00004-of-00007.safetensors", + "model.layers.19.self_attn.k_proj.bias": "model-00004-of-00007.safetensors", + "model.layers.19.self_attn.k_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.19.self_attn.o_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.19.self_attn.q_proj.bias": "model-00004-of-00007.safetensors", + "model.layers.19.self_attn.q_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.19.self_attn.v_proj.bias": "model-00004-of-00007.safetensors", + "model.layers.19.self_attn.v_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.2.input_layernorm.weight": "model-00001-of-00007.safetensors", + "model.layers.2.mlp.down_proj.weight": "model-00001-of-00007.safetensors", + "model.layers.2.mlp.gate_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.2.mlp.up_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.2.post_attention_layernorm.weight": "model-00002-of-00007.safetensors", + "model.layers.2.self_attn.k_proj.bias": "model-00002-of-00007.safetensors", + "model.layers.2.self_attn.k_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.2.self_attn.o_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.2.self_attn.q_proj.bias": "model-00002-of-00007.safetensors", + "model.layers.2.self_attn.q_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.2.self_attn.v_proj.bias": "model-00002-of-00007.safetensors", + "model.layers.2.self_attn.v_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.20.input_layernorm.weight": "model-00004-of-00007.safetensors", + "model.layers.20.mlp.down_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.20.mlp.gate_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.20.mlp.up_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.20.post_attention_layernorm.weight": "model-00004-of-00007.safetensors", + "model.layers.20.self_attn.k_proj.bias": "model-00004-of-00007.safetensors", + "model.layers.20.self_attn.k_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.20.self_attn.o_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.20.self_attn.q_proj.bias": "model-00004-of-00007.safetensors", + "model.layers.20.self_attn.q_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.20.self_attn.v_proj.bias": "model-00004-of-00007.safetensors", + "model.layers.20.self_attn.v_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.21.input_layernorm.weight": "model-00004-of-00007.safetensors", + "model.layers.21.mlp.down_proj.weight": "model-00004-of-00007.safetensors", + "model.layers.21.mlp.gate_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.21.mlp.up_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.21.post_attention_layernorm.weight": "model-00005-of-00007.safetensors", + "model.layers.21.self_attn.k_proj.bias": "model-00005-of-00007.safetensors", + "model.layers.21.self_attn.k_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.21.self_attn.o_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.21.self_attn.q_proj.bias": "model-00005-of-00007.safetensors", + "model.layers.21.self_attn.q_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.21.self_attn.v_proj.bias": "model-00005-of-00007.safetensors", + "model.layers.21.self_attn.v_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.22.input_layernorm.weight": "model-00005-of-00007.safetensors", + "model.layers.22.mlp.down_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.22.mlp.gate_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.22.mlp.up_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.22.post_attention_layernorm.weight": "model-00005-of-00007.safetensors", + "model.layers.22.self_attn.k_proj.bias": "model-00005-of-00007.safetensors", + "model.layers.22.self_attn.k_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.22.self_attn.o_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.22.self_attn.q_proj.bias": "model-00005-of-00007.safetensors", + "model.layers.22.self_attn.q_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.22.self_attn.v_proj.bias": "model-00005-of-00007.safetensors", + "model.layers.22.self_attn.v_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.23.input_layernorm.weight": "model-00005-of-00007.safetensors", + "model.layers.23.mlp.down_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.23.mlp.gate_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.23.mlp.up_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.23.post_attention_layernorm.weight": "model-00005-of-00007.safetensors", + "model.layers.23.self_attn.k_proj.bias": "model-00005-of-00007.safetensors", + "model.layers.23.self_attn.k_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.23.self_attn.o_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.23.self_attn.q_proj.bias": "model-00005-of-00007.safetensors", + "model.layers.23.self_attn.q_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.23.self_attn.v_proj.bias": "model-00005-of-00007.safetensors", + "model.layers.23.self_attn.v_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.24.input_layernorm.weight": "model-00005-of-00007.safetensors", + "model.layers.24.mlp.down_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.24.mlp.gate_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.24.mlp.up_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.24.post_attention_layernorm.weight": "model-00005-of-00007.safetensors", + "model.layers.24.self_attn.k_proj.bias": "model-00005-of-00007.safetensors", + "model.layers.24.self_attn.k_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.24.self_attn.o_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.24.self_attn.q_proj.bias": "model-00005-of-00007.safetensors", + "model.layers.24.self_attn.q_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.24.self_attn.v_proj.bias": "model-00005-of-00007.safetensors", + "model.layers.24.self_attn.v_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.25.input_layernorm.weight": "model-00005-of-00007.safetensors", + "model.layers.25.mlp.down_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.25.mlp.gate_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.25.mlp.up_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.25.post_attention_layernorm.weight": "model-00005-of-00007.safetensors", + "model.layers.25.self_attn.k_proj.bias": "model-00005-of-00007.safetensors", + "model.layers.25.self_attn.k_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.25.self_attn.o_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.25.self_attn.q_proj.bias": "model-00005-of-00007.safetensors", + "model.layers.25.self_attn.q_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.25.self_attn.v_proj.bias": "model-00005-of-00007.safetensors", + "model.layers.25.self_attn.v_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.26.input_layernorm.weight": "model-00005-of-00007.safetensors", + "model.layers.26.mlp.down_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.26.mlp.gate_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.26.mlp.up_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.26.post_attention_layernorm.weight": "model-00005-of-00007.safetensors", + "model.layers.26.self_attn.k_proj.bias": "model-00005-of-00007.safetensors", + "model.layers.26.self_attn.k_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.26.self_attn.o_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.26.self_attn.q_proj.bias": "model-00005-of-00007.safetensors", + "model.layers.26.self_attn.q_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.26.self_attn.v_proj.bias": "model-00005-of-00007.safetensors", + "model.layers.26.self_attn.v_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.27.input_layernorm.weight": "model-00005-of-00007.safetensors", + "model.layers.27.mlp.down_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.27.mlp.gate_proj.weight": "model-00005-of-00007.safetensors", + "model.layers.27.mlp.up_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.27.post_attention_layernorm.weight": "model-00006-of-00007.safetensors", + "model.layers.27.self_attn.k_proj.bias": "model-00006-of-00007.safetensors", + "model.layers.27.self_attn.k_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.27.self_attn.o_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.27.self_attn.q_proj.bias": "model-00006-of-00007.safetensors", + "model.layers.27.self_attn.q_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.27.self_attn.v_proj.bias": "model-00006-of-00007.safetensors", + "model.layers.27.self_attn.v_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.28.input_layernorm.weight": "model-00006-of-00007.safetensors", + "model.layers.28.mlp.down_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.28.mlp.gate_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.28.mlp.up_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.28.post_attention_layernorm.weight": "model-00006-of-00007.safetensors", + "model.layers.28.self_attn.k_proj.bias": "model-00006-of-00007.safetensors", + "model.layers.28.self_attn.k_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.28.self_attn.o_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.28.self_attn.q_proj.bias": "model-00006-of-00007.safetensors", + "model.layers.28.self_attn.q_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.28.self_attn.v_proj.bias": "model-00006-of-00007.safetensors", + "model.layers.28.self_attn.v_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.29.input_layernorm.weight": "model-00006-of-00007.safetensors", + "model.layers.29.mlp.down_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.29.mlp.gate_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.29.mlp.up_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.29.post_attention_layernorm.weight": "model-00006-of-00007.safetensors", + "model.layers.29.self_attn.k_proj.bias": "model-00006-of-00007.safetensors", + "model.layers.29.self_attn.k_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.29.self_attn.o_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.29.self_attn.q_proj.bias": "model-00006-of-00007.safetensors", + "model.layers.29.self_attn.q_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.29.self_attn.v_proj.bias": "model-00006-of-00007.safetensors", + "model.layers.29.self_attn.v_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.3.input_layernorm.weight": "model-00002-of-00007.safetensors", + "model.layers.3.mlp.down_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.3.mlp.gate_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.3.mlp.up_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.3.post_attention_layernorm.weight": "model-00002-of-00007.safetensors", + "model.layers.3.self_attn.k_proj.bias": "model-00002-of-00007.safetensors", + "model.layers.3.self_attn.k_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.3.self_attn.o_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.3.self_attn.q_proj.bias": "model-00002-of-00007.safetensors", + "model.layers.3.self_attn.q_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.3.self_attn.v_proj.bias": "model-00002-of-00007.safetensors", + "model.layers.3.self_attn.v_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.30.input_layernorm.weight": "model-00006-of-00007.safetensors", + "model.layers.30.mlp.down_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.30.mlp.gate_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.30.mlp.up_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.30.post_attention_layernorm.weight": "model-00006-of-00007.safetensors", + "model.layers.30.self_attn.k_proj.bias": "model-00006-of-00007.safetensors", + "model.layers.30.self_attn.k_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.30.self_attn.o_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.30.self_attn.q_proj.bias": "model-00006-of-00007.safetensors", + "model.layers.30.self_attn.q_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.30.self_attn.v_proj.bias": "model-00006-of-00007.safetensors", + "model.layers.30.self_attn.v_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.31.input_layernorm.weight": "model-00006-of-00007.safetensors", + "model.layers.31.mlp.down_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.31.mlp.gate_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.31.mlp.up_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.31.post_attention_layernorm.weight": "model-00006-of-00007.safetensors", + "model.layers.31.self_attn.k_proj.bias": "model-00006-of-00007.safetensors", + "model.layers.31.self_attn.k_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.31.self_attn.o_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.31.self_attn.q_proj.bias": "model-00006-of-00007.safetensors", + "model.layers.31.self_attn.q_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.31.self_attn.v_proj.bias": "model-00006-of-00007.safetensors", + "model.layers.31.self_attn.v_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.32.input_layernorm.weight": "model-00006-of-00007.safetensors", + "model.layers.32.mlp.down_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.32.mlp.gate_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.32.mlp.up_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.32.post_attention_layernorm.weight": "model-00006-of-00007.safetensors", + "model.layers.32.self_attn.k_proj.bias": "model-00006-of-00007.safetensors", + "model.layers.32.self_attn.k_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.32.self_attn.o_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.32.self_attn.q_proj.bias": "model-00006-of-00007.safetensors", + "model.layers.32.self_attn.q_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.32.self_attn.v_proj.bias": "model-00006-of-00007.safetensors", + "model.layers.32.self_attn.v_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.33.input_layernorm.weight": "model-00006-of-00007.safetensors", + "model.layers.33.mlp.down_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.33.mlp.gate_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.33.mlp.up_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.33.post_attention_layernorm.weight": "model-00006-of-00007.safetensors", + "model.layers.33.self_attn.k_proj.bias": "model-00006-of-00007.safetensors", + "model.layers.33.self_attn.k_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.33.self_attn.o_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.33.self_attn.q_proj.bias": "model-00006-of-00007.safetensors", + "model.layers.33.self_attn.q_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.33.self_attn.v_proj.bias": "model-00006-of-00007.safetensors", + "model.layers.33.self_attn.v_proj.weight": "model-00006-of-00007.safetensors", + "model.layers.34.input_layernorm.weight": "model-00006-of-00007.safetensors", + "model.layers.34.mlp.down_proj.weight": "model-00007-of-00007.safetensors", + "model.layers.34.mlp.gate_proj.weight": "model-00007-of-00007.safetensors", + "model.layers.34.mlp.up_proj.weight": "model-00007-of-00007.safetensors", + "model.layers.34.post_attention_layernorm.weight": "model-00007-of-00007.safetensors", + "model.layers.34.self_attn.k_proj.bias": "model-00007-of-00007.safetensors", + "model.layers.34.self_attn.k_proj.weight": "model-00007-of-00007.safetensors", + "model.layers.34.self_attn.o_proj.weight": "model-00007-of-00007.safetensors", + "model.layers.34.self_attn.q_proj.bias": "model-00007-of-00007.safetensors", + "model.layers.34.self_attn.q_proj.weight": "model-00007-of-00007.safetensors", + "model.layers.34.self_attn.v_proj.bias": "model-00007-of-00007.safetensors", + "model.layers.34.self_attn.v_proj.weight": "model-00007-of-00007.safetensors", + "model.layers.35.input_layernorm.weight": "model-00007-of-00007.safetensors", + "model.layers.35.mlp.down_proj.weight": "model-00007-of-00007.safetensors", + "model.layers.35.mlp.gate_proj.weight": "model-00007-of-00007.safetensors", + "model.layers.35.mlp.up_proj.weight": "model-00007-of-00007.safetensors", + "model.layers.35.post_attention_layernorm.weight": "model-00007-of-00007.safetensors", + "model.layers.35.self_attn.k_proj.bias": "model-00007-of-00007.safetensors", + "model.layers.35.self_attn.k_proj.weight": "model-00007-of-00007.safetensors", + "model.layers.35.self_attn.o_proj.weight": "model-00007-of-00007.safetensors", + "model.layers.35.self_attn.q_proj.bias": "model-00007-of-00007.safetensors", + "model.layers.35.self_attn.q_proj.weight": "model-00007-of-00007.safetensors", + "model.layers.35.self_attn.v_proj.bias": "model-00007-of-00007.safetensors", + "model.layers.35.self_attn.v_proj.weight": "model-00007-of-00007.safetensors", + "model.layers.4.input_layernorm.weight": "model-00002-of-00007.safetensors", + "model.layers.4.mlp.down_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.4.mlp.gate_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.4.mlp.up_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.4.post_attention_layernorm.weight": "model-00002-of-00007.safetensors", + "model.layers.4.self_attn.k_proj.bias": "model-00002-of-00007.safetensors", + "model.layers.4.self_attn.k_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.4.self_attn.o_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.4.self_attn.q_proj.bias": "model-00002-of-00007.safetensors", + "model.layers.4.self_attn.q_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.4.self_attn.v_proj.bias": "model-00002-of-00007.safetensors", + "model.layers.4.self_attn.v_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.5.input_layernorm.weight": "model-00002-of-00007.safetensors", + "model.layers.5.mlp.down_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.5.mlp.gate_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.5.mlp.up_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.5.post_attention_layernorm.weight": "model-00002-of-00007.safetensors", + "model.layers.5.self_attn.k_proj.bias": "model-00002-of-00007.safetensors", + "model.layers.5.self_attn.k_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.5.self_attn.o_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.5.self_attn.q_proj.bias": "model-00002-of-00007.safetensors", + "model.layers.5.self_attn.q_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.5.self_attn.v_proj.bias": "model-00002-of-00007.safetensors", + "model.layers.5.self_attn.v_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.6.input_layernorm.weight": "model-00002-of-00007.safetensors", + "model.layers.6.mlp.down_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.6.mlp.gate_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.6.mlp.up_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.6.post_attention_layernorm.weight": "model-00002-of-00007.safetensors", + "model.layers.6.self_attn.k_proj.bias": "model-00002-of-00007.safetensors", + "model.layers.6.self_attn.k_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.6.self_attn.o_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.6.self_attn.q_proj.bias": "model-00002-of-00007.safetensors", + "model.layers.6.self_attn.q_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.6.self_attn.v_proj.bias": "model-00002-of-00007.safetensors", + "model.layers.6.self_attn.v_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.7.input_layernorm.weight": "model-00002-of-00007.safetensors", + "model.layers.7.mlp.down_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.7.mlp.gate_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.7.mlp.up_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.7.post_attention_layernorm.weight": "model-00002-of-00007.safetensors", + "model.layers.7.self_attn.k_proj.bias": "model-00002-of-00007.safetensors", + "model.layers.7.self_attn.k_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.7.self_attn.o_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.7.self_attn.q_proj.bias": "model-00002-of-00007.safetensors", + "model.layers.7.self_attn.q_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.7.self_attn.v_proj.bias": "model-00002-of-00007.safetensors", + "model.layers.7.self_attn.v_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.8.input_layernorm.weight": "model-00002-of-00007.safetensors", + "model.layers.8.mlp.down_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.8.mlp.gate_proj.weight": "model-00002-of-00007.safetensors", + "model.layers.8.mlp.up_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.8.post_attention_layernorm.weight": "model-00003-of-00007.safetensors", + "model.layers.8.self_attn.k_proj.bias": "model-00003-of-00007.safetensors", + "model.layers.8.self_attn.k_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.8.self_attn.o_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.8.self_attn.q_proj.bias": "model-00003-of-00007.safetensors", + "model.layers.8.self_attn.q_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.8.self_attn.v_proj.bias": "model-00003-of-00007.safetensors", + "model.layers.8.self_attn.v_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.9.input_layernorm.weight": "model-00003-of-00007.safetensors", + "model.layers.9.mlp.down_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.9.mlp.gate_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.9.mlp.up_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.9.post_attention_layernorm.weight": "model-00003-of-00007.safetensors", + "model.layers.9.self_attn.k_proj.bias": "model-00003-of-00007.safetensors", + "model.layers.9.self_attn.k_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.9.self_attn.o_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.9.self_attn.q_proj.bias": "model-00003-of-00007.safetensors", + "model.layers.9.self_attn.q_proj.weight": "model-00003-of-00007.safetensors", + "model.layers.9.self_attn.v_proj.bias": "model-00003-of-00007.safetensors", + "model.layers.9.self_attn.v_proj.weight": "model-00003-of-00007.safetensors", + "model.norm.weight": "model-00007-of-00007.safetensors" + } +} diff --git a/pictures/Abstrct.png b/pictures/Abstrct.png new file mode 100644 index 0000000..3c3cc37 --- /dev/null +++ b/pictures/Abstrct.png @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6c800e55c302c62ef138726a1db09fa5d3a7c60e964b4769cedc491bc0dd7254 +size 215002 diff --git a/pictures/Acc_and_Scale.png b/pictures/Acc_and_Scale.png new file mode 100644 index 0000000..311e8d2 --- /dev/null +++ b/pictures/Acc_and_Scale.png @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e05d10eb97ec6044e03cf4ce1c7eeb94f392e618049490700ea26c994b6978a9 +size 103916 diff --git a/pictures/Architecture.png b/pictures/Architecture.png new file mode 100644 index 0000000..d11fdbc --- /dev/null +++ b/pictures/Architecture.png @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:71f5f02ff71fe4b47be25c94e0e7dceeb813ce8baf7f3f69d93a613b11ede15e +size 125754 diff --git a/pictures/LeetCode.png b/pictures/LeetCode.png new file mode 100644 index 0000000..1ef5626 --- /dev/null +++ b/pictures/LeetCode.png @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8d7f23be2515c6be4228983e48697b90a89de0f1b5dbc636e64872f164176a59 +size 228922 diff --git a/pictures/VibeThiinker-3B.png b/pictures/VibeThiinker-3B.png new file mode 100644 index 0000000..e79854c --- /dev/null +++ b/pictures/VibeThiinker-3B.png @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4ae18631f39b472cfab1f327c659a4ea2c4d111b6e739fc4764709cfbaa49a06 +size 248723 diff --git a/pictures/VibeThinker-3B+CLR.png b/pictures/VibeThinker-3B+CLR.png new file mode 100644 index 0000000..dd95439 --- /dev/null +++ b/pictures/VibeThinker-3B+CLR.png @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9ce66aa2fb3c5c93f1b8c7746c65d2217aedce19218d96f10b16092c0bfe0508 +size 213442 diff --git a/pictures/animation-VibeThinker-3B.gif b/pictures/animation-VibeThinker-3B.gif new file mode 100644 index 0000000..70573d6 --- /dev/null +++ b/pictures/animation-VibeThinker-3B.gif @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5451de26ea2fe763a8240bdee3ea4a8d4e89b083aea247b274887605918cd987 +size 11715037 diff --git a/tokenizer.json b/tokenizer.json new file mode 100644 index 0000000..c4c8048 --- /dev/null +++ b/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:287f2606720ee3cee69948de373ad0f1a741765bbd4a729d27e0c2b4037bd206 +size 11422430 diff --git a/tokenizer_config.json b/tokenizer_config.json new file mode 100644 index 0000000..86e625b --- /dev/null +++ b/tokenizer_config.json @@ -0,0 +1,16 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "is_local": true, + "local_files_only": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "padding_side": "left", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +}