From cfab229d8da692df5b1a7c72b61490e143d75e39 Mon Sep 17 00:00:00 2001 From: ModelHub XC Date: Mon, 20 Jul 2026 16:38:09 +0800 Subject: [PATCH] =?UTF-8?q?=E5=88=9D=E5=A7=8B=E5=8C=96=E9=A1=B9=E7=9B=AE?= =?UTF-8?q?=EF=BC=8C=E7=94=B1ModelHub=20XC=E7=A4=BE=E5=8C=BA=E6=8F=90?= =?UTF-8?q?=E4=BE=9B=E6=A8=A1=E5=9E=8B?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Model: ArcOffical/PiCo-1B Source: Original Platform --- .gitattributes | 36 +++++ LICENSE | 302 +++++++++++++++++++++++++++++++++++++++++ README.md | 177 ++++++++++++++++++++++++ chat_template.jinja | 54 ++++++++ config.json | 62 +++++++++ generation_config.json | 7 + model.safetensors | 3 + tokenizer.json | 3 + tokenizer_config.json | 30 ++++ 9 files changed, 674 insertions(+) create mode 100644 .gitattributes create mode 100644 LICENSE create mode 100644 README.md create mode 100644 chat_template.jinja create mode 100644 config.json create mode 100644 generation_config.json create mode 100644 model.safetensors create mode 100644 tokenizer.json create mode 100644 tokenizer_config.json diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..52373fe --- /dev/null +++ b/.gitattributes @@ -0,0 +1,36 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text +tokenizer.json filter=lfs diff=lfs merge=lfs -text diff --git a/LICENSE b/LICENSE new file mode 100644 index 0000000..9eba358 --- /dev/null +++ b/LICENSE @@ -0,0 +1,302 @@ +PiCo-1B Open RAIL-M License +Version 1.0, dated June 25, 2026 + + +Section I: PREAMBLE + +This Open RAIL-M License Agreement was created for PiCo-1B, a machine-learning +model developed by ArcOffical. This license is generally applicable to any +machine-learning Model. + +This License Agreement strives for both the open and responsible Use of the +accompanying Model. Openness here is understood as enabling users of the Model +on a royalty-free basis to Use it, modify it, and even share commercial versions +of it. Use restrictions are included to prevent misuse of the Model. + +This License Agreement governs the Use of the Model and Modifications of the +Model. Even though downstream derivative versions of the model could be released +under different licensing terms, the latter will always have to include — at +minimum — the same use-based restrictions as the ones in the original license. + +The development and use of artificial intelligence ("AI") does not come without +concerns. The world has witnessed how AI techniques may, in some instances, +become risky for the public in general. These risks come in many forms, from +racial discrimination to the misuse of sensitive information. ArcOffical +believes in the intersection between open and responsible AI development; thus, +this License aims to strike a balance between both in order to enable +responsible open-science in the field of AI. + +This License governs the use of the Model (and its Derivatives) and is informed +by the model card associated with the Model. + +NOW THEREFORE, You and Licensor agree as follows: + + +Section II: DEFINITIONS + +1.1 "Complementary Material" means the accompanying source code and scripts used +to define, run, load, benchmark or evaluate the Model, and used to prepare data +for training or evaluation, if any. This includes any accompanying +documentation, tutorials, examples, etc., if any. + +1.2 "Contribution" means any work of authorship, including the original version +of the Model and any Modifications or additions to that Model or Derivatives of +the Model thereof, that is intentionally submitted to Licensor for inclusion in +the Model by the copyright owner or by an individual or legal entity authorized +to submit on behalf of the copyright owner. For the purposes of this definition, +"submitted" means any form of electronic, verbal, or written communication sent +to the Licensor or its representatives, including but not limited to +communication on electronic mailing lists, source code control systems, and +issue tracking systems that are managed by, or on behalf of, the Licensor for +the purpose of discussing and improving the Model, but excluding communication +that is conspicuously marked or otherwise designated in writing by the copyright +owner as "Not a Contribution". + +1.3 "Contributor" means Licensor and any individual or legal entity on behalf of +whom a Contribution has been received by Licensor and subsequently incorporated +within the Model. + +1.4 "Data" means a collection of information and/or content extracted from the +dataset used with the Model, including to train, pretrain, or otherwise evaluate +the Model. The Data is not licensed under this License. + +1.5 "Derivatives of the Model" means all modifications to the Model, works based +on the Model, or any other model which is created or initialized by transfer of +patterns of the weights, parameters, activations or output of the Model, to the +other model, in order to cause the other model to perform similarly to the +Model, including — but not limited to — distillation methods entailing the use +of intermediate data representations or methods based on the generation of +synthetic data by the Model for training the other model. + +1.6 "Distribution" means any transmission, reproduction, publication or other +sharing of the Model or Derivatives of the Model to a third party, including +providing the Model as a hosted or cloud service made available by electronic or +other remote means — e.g., API-based or web-access. + +1.7 "Harm" includes but is not limited to physical, mental, psychological, +financial and reputational damage, pain, or loss. + +1.8 "License" means the terms and conditions for use, reproduction, and +Distribution as defined in this document. + +1.9 "Licensor" means ArcOffical, the rights owner granting the terms and +conditions of this License Agreement. + +1.10 "Model" means the PiCo-1B machine-learning based assemblies (including +checkpoints), consisting of learnt weights, parameters (including optimizer +states), corresponding to the model architecture as embodied in the +Complementary Material, that have been trained or tuned, in whole or in part on +the Data, using the Complementary Material. + +1.11 "Output" means the results of operating a Model as embodied in +informational content resulting therefrom. + +1.12 "You" (or "Your") means an individual or legal entity exercising +permissions granted by this License and/or making use of the Model for whichever +purpose and in any field of use, including commercial usage of the Model in an +end-use application. + + +Section III: GRANT OF RIGHTS + +3.1 Grant of License. Subject to Your compliance with all terms and conditions +of this License, Licensor hereby grants You a worldwide, royalty-free, +non-exclusive, perpetual (for the duration of the applicable copyright) license +to: + +(a) Use, reproduce, distribute, and publicly perform the Model; + +(b) Create, use, reproduce, distribute, and publicly perform Derivatives of the +Model; + +(c) Sublicense the Model and Derivatives of the Model to third parties, provided +that any such sublicense includes all terms and conditions of this License and +the same use-based restrictions; + +(d) Use the Model and Derivatives of the Model for commercial purposes, +including but not limited to: + - Offering the Model as a hosted or API service; + - Integrating the Model into commercial products or services; + - Selling or licensing Derivatives of the Model. + +3.2 Reservation of Rights. All rights not expressly granted to You under this +License are reserved by Licensor. Nothing in this License constitutes a transfer +of ownership of the Model or any intellectual property rights therein. + + +Section IV: USE-BASED RESTRICTIONS + +You may not Use, reproduce, distribute, or create Derivatives of the Model for +any purpose that: + +4.1 Violates Law or Human Rights. + - Violates any applicable national or international law, regulation, or + treaty; + - Violates internationally recognized human rights, including but not limited + to the Universal Declaration of Human Rights. + +4.2 Generates or Promotes Harmful Content. + - Generate, promote, or facilitate the generation of content that is + defamatory, libelous, obscene, pornographic, or otherwise objectionable; + - Generate, promote, or facilitate the generation of content that incites, + promotes, or glorifies violence, terrorism, or criminal activities; + - Generate child sexual abuse material (CSAM) or content that sexualizes + minors; + - Generate content that promotes suicide, self-harm, or eating disorders; + - Generate content that promotes illegal drugs, gambling addiction, or other + addictive or harmful behaviors. + +4.3 Discriminates or Harms. + - Generate, promote, or facilitate the generation of content that + discriminates against, harasses, or disparages individuals or groups on the + basis of race, color, religion, gender, gender identity, sexual + orientation, national origin, age, disability, or any other protected + characteristic; + - Generate content that promotes hate speech or incites hatred against any + individual or group. + +4.4 Violates Privacy or Enables Surveillance. + - Generate deepfakes, impersonations, or other synthetic media without the + explicit informed consent of the individuals depicted; + - Identify, track, or surveil individuals without their explicit informed + consent, including but not limited to facial recognition, biometric + identification, or location tracking; + - Generate content for the purpose of identity theft, fraud, or deception; + - Generate content for the purpose of spreading disinformation, + misinformation, or propaganda. + +4.5 Provides Professional Advice. + - Provide or replace professional medical diagnosis, treatment, or advice; + - Provide or replace professional legal advice; + - Provide or replace professional financial, investment, or tax advice. + +4.6 Automates Harmful Decision-Making. + - Develop automated decision-making systems that have a significant impact on + individuals' legal status, employment, housing, credit, education, or other + major life opportunities, without meaningful human oversight and the + ability for individuals to appeal automated decisions; + - Develop systems for use in law enforcement, criminal justice, or national + security contexts without appropriate safeguards and oversight. + +4.7 Evades or Circumvents Restrictions. + - Evade, circumvent, or attempt to evade or circumvent any of the use-based + restrictions set forth in this Section IV, whether through the creation of + Derivatives, the use of Output, or any other means. + + +Section V: YOUR OBLIGATIONS + +5.1 Retention of License and Notices. + - You must retain, in all copies of the Model and Derivatives of the Model, + this License Agreement in its entirety, including all copyright and other + proprietary notices. + - You must include a copy of this License Agreement with any Distribution of + the Model or Derivatives of the Model. + +5.2 Application of Use-Based Restrictions to Derivatives. + - Any Derivatives of the Model that You create, distribute, or sublicense + must be subject to — at a minimum — the same use-based restrictions as set + forth in Section IV of this License. + - You may add additional restrictions to Your Derivatives, provided that such + additional restrictions do not diminish, remove, or conflict with the + use-based restrictions in Section IV. + +5.3 Attribution. + - Any public use, publication, or distribution of Output generated by the + Model or Derivatives of the Model must include clear attribution to + "PiCo-1B by ArcOffical". + - Such attribution must state that the Output is AI-generated and may contain + errors or inaccuracies. + - You may not use the name "ArcOffical", "PiCo", or "PiCo-1B" to endorse or + promote any product, service, or derivative work without prior written + permission from Licensor. + +5.4 Model Card and Documentation. + - If You Distribute the Model or Derivatives of the Model, You are strongly + encouraged to provide or maintain a model card or similar documentation + that describes the capabilities, limitations, and appropriate use cases of + Your Model or Derivatives. + +5.5 Reporting Violations. + - If You become aware of any Use of the Model or Derivatives of the Model + that violates this License, You should promptly notify Licensor at + [INSERT CONTACT EMAIL]. + + +Section VI: INTELLECTUAL PROPERTY + +6.1 Ownership. Licensor retains all right, title, and interest in and to the +Model and any intellectual property rights therein. Contributors retain all +right, title, and interest in and to their Contributions. + +6.2 Trademark. This License does not grant You any rights to use the trademarks, +service marks, or logos of Licensor or any Contributor, except as required for +reasonable and customary use in describing the origin of the Model and +reproducing the content of the copyright notice. + + +Section VII: DISCLAIMER OF WARRANTY + +THE MODEL AND COMPLEMENTARY MATERIAL ARE PROVIDED "AS IS", WITHOUT WARRANTY OF +ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF +MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE, AND NONINFRINGEMENT. LICENSOR +AND CONTRIBUTORS DO NOT WARRANT THAT THE MODEL WILL MEET YOUR REQUIREMENTS OR +THAT THE OPERATION OF THE MODEL WILL BE UNINTERRUPTED OR ERROR-FREE. + + +Section VIII: LIMITATION OF LIABILITY + +IN NO EVENT SHALL LICENSOR OR CONTRIBUTORS BE LIABLE FOR ANY CLAIM, DAMAGES, OR +OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT, OR OTHERWISE, ARISING +FROM, OUT OF, OR IN CONNECTION WITH THE MODEL, COMPLEMENTARY MATERIAL, OR THE +USE OR OTHER DEALINGS IN THE MODEL. + + +Section IX: TERMINATION + +9.1 Automatic Termination. This License and the rights granted hereunder will +terminate automatically upon any breach by You of the terms of this License. + +9.2 Survival. Sections I (Preamble), II (Definitions), VI (Intellectual +Property), VII (Disclaimer of Warranty), VIII (Limitation of Liability), IX +(Termination), and X (General Provisions) shall survive any termination of this +License. + +9.3 Termination of Derivative Licenses. If You terminate this License, any +sublicenses granted by You to third parties prior to termination shall survive +termination, provided that such third parties remain in full compliance with all +terms of this License. + + +Section X: GENERAL PROVISIONS + +10.1 Entire Agreement. This License constitutes the entire agreement between You +and Licensor regarding the subject matter hereof and supersedes all prior or +contemporaneous communications, whether written or oral. + +10.2 Severability. If any provision of this License is held to be invalid, +illegal, or unenforceable, the remaining provisions shall remain in full force +and effect, and the invalid provision shall be reformed to the minimum extent +necessary to make it valid and enforceable. + +10.3 No Waiver. The failure of Licensor to enforce any provision of this License +shall not constitute a waiver of such provision or of the right to enforce such +provision. + +10.4 Governing Law. This License shall be governed by and construed in +accordance with the laws of [INSERT JURISDICTION], without regard to its +conflict of laws principles. + +10.5 Dispute Resolution. Any dispute arising out of or relating to this License +shall be resolved through [INSERT DISPUTE RESOLUTION MECHANISM, e.g., binding +arbitration in accordance with the rules of the ICC / exclusive jurisdiction of +the courts of JURISDICTION]. + + +Section XI: CONTACT INFORMATION + +For inquiries regarding this License, please contact: + +ArcOffical +[yuetwuism6home@gmail.com] +[https://huggingface.co/ArcOffical/PiCo-1B/tree/main] diff --git a/README.md b/README.md new file mode 100644 index 0000000..96a6b43 --- /dev/null +++ b/README.md @@ -0,0 +1,177 @@ +--- +license: openrail +language: +- en +pipeline_tag: text-generation +--- +# PiCo 1B + +> A 1B-parameter dense language model optimized for reasoning and knowledge tasks. +> +> For clarity, our model uses the tokenizer from Qwen 2 1.5B but has been trained from scratch — it is not a fine-tuned version of Qwen 2 1.5B. + +--- + +## 📌 Model Overview + +**PiCo 1B** is a compact, high-performance language model with ~1.46 billion parameters. Despite its small size, it achieves competitive performance across reasoning, knowledge, and coding benchmarks, particularly excelling in science reasoning tasks. + +--- + +## 📋 Model Details + +| Attribute | Value | +|-----------|-------| +| **Model Size** | ~1.46B parameters | +| **Architecture** | Dense transformer (decoder-only) | +| **Context Length** | 2048 tokens | +| **Precision** | FP32 / FP16 / Safetensors | +| **License** | Open-source | + +--- + +## 📊 Benchmark Results + +PiCo 1B is evaluated against **31 open-source models** in the 1B–2B parameter range across 7 standard benchmarks. + +### MMLU (Massive Multitask Language Understanding) + +Measures general knowledge across 57 subjects including STEM, humanities, and social sciences. + +![mmlu_comparison](https://cdn-uploads.huggingface.co/production/uploads/67f51c8931456897b8918a16/xRmbn3pMYD6hLVEPZYwBa.png) + +--- + +### GSM8K (Grade School Math) + +Measures mathematical reasoning with grade-school level word problems. + +![gsm8k_comparison](https://cdn-uploads.huggingface.co/production/uploads/67f51c8931456897b8918a16/E03-rJ-NpEwCH5M97ec-v.png) + +--- + +### ARC-Challenge (AI2 Reasoning Challenge) + +Measures science reasoning with grade-level science questions (harder subset). + +![arc_challenge_comparison](https://cdn-uploads.huggingface.co/production/uploads/67f51c8931456897b8918a16/WOOinEwZ0DaRKJl74n6E9.png) + +--- + +### ARC-Easy (AI2 Reasoning Challenge) + +Measures basic science reasoning with grade-level science questions (easier subset). + +![arc_easy_comparison](https://cdn-uploads.huggingface.co/production/uploads/67f51c8931456897b8918a16/BTGhsnWy0gA6iJaHBr8j4.png) + +--- + +### HellaSwag (Commonsense Reasoning) + +Measures commonsense natural language inference with everyday scenarios. + +![hellaswag_comparison](https://cdn-uploads.huggingface.co/production/uploads/67f51c8931456897b8918a16/pNLjRk1Xy6aoGFOik3Zsv.png) + +--- + +### HumanEval (Code Generation) + +Measures functional correctness of code generation across 164 programming problems. + +![humaneval_comparison](https://cdn-uploads.huggingface.co/production/uploads/67f51c8931456897b8918a16/FnhNO9sBCDl-Ky438q6ll.png) + +--- + +### TruthfulQA (Truthfulness) + +Measures whether the model generates truthful answers rather than mimicking common misconceptions. + +![truthfulqa_comparison](https://cdn-uploads.huggingface.co/production/uploads/67f51c8931456897b8918a16/pKq1fhlzmfistFhZ8BRRt.png) + +--- + +## 🏆 Performance Highlights + +### ✅ Strengths + +- **Science Reasoning**: Best-in-class performance on ARC-Easy and ARC-Challenge +- **General Knowledge**: Top 3 on MMLU, outperforming many larger 1.5B–2B models +- **Coding Ability**: Strong HumanEval performance, competitive with models 2x its size +- **Truthfulness**: Top 5 on TruthfulQA, demonstrating reliable factual output + +### 📈 Areas for Improvement + +- **Commonsense Reasoning**: HellaSwag score lags behind modern 1.5B+ models +- **Mathematical Reasoning**: GSM8K performance is solid but not top-tier +- **Scale**: Further training on larger, more diverse datasets could boost all benchmarks + +--- + +## 🚀 Usage + +### Quick Start + +```python +from transformers import AutoModelForCausalLM, AutoTokenizer + +model_name = "pico-1b" +tokenizer = AutoTokenizer.from_pretrained(model_name) +model = AutoModelForCausalLM.from_pretrained(model_name) + +prompt = "Explain the theory of relativity in simple terms." +inputs = tokenizer(prompt, return_tensors="pt") +outputs = model.generate(**inputs, max_length=200) +print(tokenizer.decode(outputs[0], skip_special_tokens=True)) +``` + +### Model Formats + +- **Safetensors** (recommended): Secure and fast loading +- **PyTorch (FP16)**: Standard format +- **GGUF**: For local inference with llama.cpp + +--- + +## 🏋️ Training Details + +| Aspect | Description | +|--------|-------------| +| **Architecture** | Dense decoder-only transformer | +| **Optimizer** | AdamW | +| **Learning Rate** | Cosine schedule with warmup | +| **Batch Size** | Configurable per GPU setup | +| **Training Framework** | PyTorch + Hugging Face Transformers | + +--- + +## ⚠️ Limitations + +- **Small Model Size**: As a 1B-parameter model, it has inherent limitations compared to larger models (7B+) on complex reasoning tasks +- **Training Data**: Primarily trained on English text; performance on non-English languages may be limited +- **Hallucinations**: Like all LLMs, it may generate factually incorrect information +- **Context Window**: Limited to 2048 tokens by default + +--- + +## 📝 Citation + +If you use PiCo 1B in your research or projects, please cite: + +```bibtex +@misc{pico1b, + title={PiCo 1B: A Compact Language Model Optimized for Reasoning}, + author={Arc Develop Team}, + year={2026}, + howpublished={\url{https://github.com/pico-llm/pico-1b}}, +} +``` + +--- + +## 📄 License + +This model is released under an open-source license. Please see the LICENSE file for details. + +--- + +*Last updated: June 2026* \ No newline at end of file diff --git a/chat_template.jinja b/chat_template.jinja new file mode 100644 index 0000000..28028c0 --- /dev/null +++ b/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/config.json b/config.json new file mode 100644 index 0000000..7b0937c --- /dev/null +++ b/config.json @@ -0,0 +1,62 @@ +{ + "architectures": [ + "Qwen2ForCausalLM" + ], + "attention_dropout": 0.0, + "bos_token_id": 151643, + "dtype": "bfloat16", + "eos_token_id": 151643, + "hidden_act": "silu", + "hidden_size": 1536, + "initializer_range": 0.02, + "intermediate_size": 8960, + "layer_types": [ + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention" + ], + "max_position_embeddings": 131072, + "max_window_layers": 28, + "model_type": "qwen2", + "num_attention_heads": 12, + "num_hidden_layers": 28, + "num_key_value_heads": 2, + "pad_token_id": null, + "rms_norm_eps": 1e-06, + "rope_parameters": { + "rope_theta": 1000000.0, + "rope_type": "default" + }, + "sliding_window": null, + "tie_word_embeddings": true, + "transformers_version": "5.12.1", + "use_cache": true, + "use_mrope": false, + "use_sliding_window": false, + "vocab_size": 151936 +} diff --git a/generation_config.json b/generation_config.json new file mode 100644 index 0000000..4639464 --- /dev/null +++ b/generation_config.json @@ -0,0 +1,7 @@ +{ + "bos_token_id": 151643, + "do_sample": false, + "eos_token_id": 151643, + "max_new_tokens": 2048, + "transformers_version": "5.12.1" +} diff --git a/model.safetensors b/model.safetensors new file mode 100644 index 0000000..a74e359 --- /dev/null +++ b/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bcf18ee39c3575a4c8df94713dbde01a8ee5bc5a0404d254944f66c6d90a1b33 +size 3087467144 diff --git a/tokenizer.json b/tokenizer.json new file mode 100644 index 0000000..34510ff --- /dev/null +++ b/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3fd169731d2cbde95e10bf356d66d5997fd885dd8dbb6fb4684da3f23b2585d8 +size 11421892 diff --git a/tokenizer_config.json b/tokenizer_config.json new file mode 100644 index 0000000..df536e9 --- /dev/null +++ b/tokenizer_config.json @@ -0,0 +1,30 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "local_files_only": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +}