commit 0556e549f120d75a5af6cf6f8672e4662cd01c9f Author: ModelHub XC Date: Fri Jul 24 05:23:11 2026 +0800 初始化项目,由ModelHub XC社区提供模型 Model: JuliaKreutzerCohere/tiny-aya-global-prompt-tasktype Source: Original Platform diff --git a/.eval_results/gpqa-diamond.yaml b/.eval_results/gpqa-diamond.yaml new file mode 100644 index 0000000..122fc4f --- /dev/null +++ b/.eval_results/gpqa-diamond.yaml @@ -0,0 +1,9 @@ +- dataset: + id: Idavidrein/gpqa + task_id: diamond + date: '2026-04-18' + notes: GPQA Diamond + source: + name: EvalEval + url: https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/0b/5b/0b5bf7f7-df84-4ef0-a46e-b2de6eab325c.json + value: 28.2828282828 diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..771fc53 --- /dev/null +++ b/.gitattributes @@ -0,0 +1,51 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text +tokenizer.json filter=lfs diff=lfs merge=lfs -text +assets/TinyAya_Global.png filter=lfs diff=lfs merge=lfs -text +wheels/certifi-2026.6.17-py3-none-any.whl filter=lfs diff=lfs merge=lfs -text +wheels/charset_normalizer-3.4.9-cp310-cp310-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl filter=lfs diff=lfs merge=lfs -text +wheels/fsspec-2026.6.0-py3-none-any.whl filter=lfs diff=lfs merge=lfs -text +wheels/hf_xet-1.5.1-cp37-abi3-manylinux2014_x86_64.manylinux_2_17_x86_64.whl filter=lfs diff=lfs merge=lfs -text +wheels/huggingface_hub-0.36.2-py3-none-any.whl filter=lfs diff=lfs merge=lfs -text +wheels/numpy-2.2.6-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl filter=lfs diff=lfs merge=lfs -text +wheels/packaging-26.2-py3-none-any.whl filter=lfs diff=lfs merge=lfs -text +wheels/pyyaml-6.0.3-cp310-cp310-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl filter=lfs diff=lfs merge=lfs -text +wheels/regex-2026.7.10-cp310-cp310-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl filter=lfs diff=lfs merge=lfs -text +wheels/safetensors-0.8.0-cp310-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl filter=lfs diff=lfs merge=lfs -text +wheels/tokenizers-0.22.2-cp39-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl filter=lfs diff=lfs merge=lfs -text +wheels/tqdm-4.68.4-py3-none-any.whl filter=lfs diff=lfs merge=lfs -text +wheels/transformers-4.56.2-py3-none-any.whl filter=lfs diff=lfs merge=lfs -text +wheels/urllib3-2.7.0-py3-none-any.whl filter=lfs diff=lfs merge=lfs -text diff --git a/README.md b/README.md new file mode 100644 index 0000000..5d503ec --- /dev/null +++ b/README.md @@ -0,0 +1,240 @@ +--- +inference: false +library_name: transformers +language: +- en +- nl +- fr +- it +- pt +- ro +- es +- cs +- pl +- uk +- ru +- el +- de +- da +- sv +- "no" +- ca +- gl +- cy +- ga +- eu +- hr +- lv +- lt +- sk +- sl +- et +- fi +- hu +- sr +- bg +- ar +- fa +- ur +- tr +- mt +- he +- hi +- mr +- bn +- gu +- pa +- ta +- te +- ne +- tl +- ms +- id +- vi +- jv +- km +- th +- lo +- zh +- my +- ja +- ko +- am +- ha +- ig +- mg +- sn +- sw +- wo +- xh +- yo +- zu +license: cc-by-nc-4.0 +extra_gated_prompt: >- + By submitting this form, you agree to the [License + Agreement](https://cohere.com/c4ai-cc-by-nc-license) and acknowledge that the + information you provide will be collected, used, and shared in accordance with + Cohere's [Privacy Policy]( https://cohere.com/privacy). You'll receive email + updates about Cohere Labs and Cohere research, events, products and services. + You can unsubscribe at any time. +extra_gated_fields: + Name: text + Affiliation: text + Country: country + I agree to use this model for non-commercial use ONLY: checkbox +base_model: CohereLabs/tiny-aya-base +--- + +# **Model Card for tiny-aya-global** + +![Tiny Aya Global](./assets/TinyAya_Global.png) + +**Best balance across languages and regions.** For other regions, check [tiny-aya-fire](https://huggingface.co/CohereLabs/tiny-aya-fire), [tiny-aya-earth](https://huggingface.co/CohereLabs/tiny-aya-earth), [tiny-aya-water](https://huggingface.co/CohereLabs/tiny-aya-water) + +## **Model Summary** + +Cohere Labs Tiny Aya is an open weights research release of a pretrained 3.35 billion parameter model optimized for efficient, strong, and balanced multilingual representation across 70+ languages, including many lower-resourced ones. The model is designed to support downstream adaptation, instruction tuning, and local deployment under realistic compute constraints. + +Developed by: [Cohere](https://cohere.com/) and [Cohere](https://cohere.com/research) Labs + +* Point of Contact: [**Cohere Labs**](https://cohere.com/research) +* License: [CC-BY-NC](https://cohere.com/cohere-labs-cc-by-nc-license), requires also adhering to **[Cohere Lab's Acceptable Use Policy](https://docs.cohere.com/docs/c4ai-acceptable-use-policy)** +* Model: tiny-aya-it-global +* Model Size: 3.35B +* Context length: 8K input + +For more details about this model family, please check out our [blog post](https://cohere.com/blog/cohere-labs-tiny-aya) and [tech report](https://arxiv.org/abs/2603.11510). + +**Try Cohere Labs Tiny Aya** + +You can try out Cohere Labs Tiny Aya before downloading the weights in our hosted [Hugging Face Space](https://huggingface.co/spaces/CohereLabs/tiny-aya). + +**Usage** + +```py +from transformers import AutoTokenizer, AutoModelForCausalLM + +model_id = "CohereLabs/tiny-aya-global" +tokenizer = AutoTokenizer.from_pretrained(model_id) +model = AutoModelForCausalLM.from_pretrained(model_id) + +# Format message with the chat template +messages = [{"role": "user", "content": "Explica en español qué significa la palabra japonesa 'ikigai' y da un ejemplo práctico."}] +input_ids = tokenizer.apply_chat_template( + messages, + tokenize=True, + add_generation_prompt=True, + return_tensors="pt", +) + +gen_tokens = model.generate( + input_ids, + max_new_tokens=4096, + do_sample=True, + temperature=0.1, + top_p=0.95 +) + +gen_text = tokenizer.decode(gen_tokens[0]) +print(gen_text) +``` + +You can also use the model directly using transformers `pipeline` abstraction: + +```py +from transformers import pipeline +import torch + +model_id = "CohereLabs/tiny-aya-global" + +pipe = pipeline( + "text-generation", + model=model_id, + torch_dtype="auto", + device_map="auto", +) + +messages = [ + {"role": "user", "content": "Explain the Transformer architecture"}, +] + +text = tokenizer.apply_chat_template( + messages, + tokenize=False, + add_generation_prompt=True, +) + + +outputs = pipe( + messages, + max_new_tokens=300, +) +print(outputs[0]["generated_text"][-1]) + +``` + +## **Model Details** + +**Input**: Text only. + +**Output**: Model generates text. + +**Model Architecture**: This is an auto-regressive language model that uses an optimized transformer architecture. After pretraining, this model uses supervised fine-tuning (SFT) and preference training to align model behavior to human preferences for helpfulness and safety. The model features three layers with sliding window attention (window size 4096\) and RoPE for efficient local context modeling and relative positional encoding. A fourth layer uses global attention without positional embeddings, enabling unrestricted token interactions across the entire sequence. + +**Languages covered:** The model has been trained on 70+ languages, with a focus on: English, Dutch, French, Italian, Portuguese, Romanian, Spanish, Czech, Polish, Ukrainian, Russian, Greek, German, Danish, Swedish, Norwegian, Catalan, Galician, Welsh, Irish, Basque, Croatian, Latvian, Lithuanian, Slovak, Slovenian, Estonian, Finnish, Hungarian, Serbian, Bulgarian, Arabic, Persian, Urdu, Turkish, Maltese, Hebrew, Hindi, Marathi, Bengali, Gujarati, Punjabi, Tamil, Telugu, Nepali, Tagalog, Malay, Indonesian, Vietnamese, Javanese, Khmer, Thai, Lao, Chinese, Burmese, Japanese, Korean, Amharic, Hausa, Igbo, Malagasy, Shona, Swahili, Wolof, Xhosa, Yoruba, and Zulu + +**Context Length:** Tiny Aya supports a context length of 8K & 8K output length. + +![Language Understanding on CC](./assets/tiny-aya-lowres-dotplot_lightmode.png) + +![Performance Comparison](./assets/TinyAya_PlotB_v7_lightmode.png) + +![Generation Quality on Dolly](./assets/TinyAyaPlot_D_Light.png) + +## **Usage and Limitations** + +### **Intended Usage** + +Tiny Aya is a family of massively multilingual small language models built to bring capable AI to languages that are often underserved by existing models. The models support languages across Indic, East and Southeast Asian, African, European, and Middle Eastern language families, with a deliberate emphasis on low-resource language performance. + +Intended applications include multilingual text generation, conversational AI, summarization, translation and cross-lingual tasks, as well as research in multilingual NLP and low-resource language modeling. The models are also suited for efficient deployment in multilingual regions, helping bridge the digital language divide for underrepresented language communities. + +### **Strengths** + +Tiny Aya demonstrates strong open-ended generation quality across its full language coverage, with particularly notable performance on low-resource languages. The model performs well on translation, summarization, and cross-lingual tasks, benefiting from training signal shared across language families and scripts. + +### **Limitations** + +**Reasoning tasks.** The model's strongest performance is on open-ended generation and conversational tasks. Chain-of-thought reasoning tasks such as multilingual math (MGSM) are comparatively weaker. + +**Factual knowledge.** As with any language model, outputs may contain incorrect or outdated statements, particularly in lower-resource languages with thinner training data coverage. + +**Uneven resource distribution.** High-resource languages benefit from richer training signal and tend to exhibit more consistent quality across tasks. The lowest-resource languages in the model's coverage may show greater variability, and culturally specific nuance, sarcasm, or figurative language may be less reliably handled in these languages. + +**Task complexity.** The model performs best with clear prompts and instructions. Highly complex or open-ended reasoning, particularly in lower-resource languages, remains challenging. + +## **Model Card Contact** + +For errors or additional questions about details in this model card, contact \[labs@cohere.com\]. + +## **Terms of Use:** + +We hope that the release of this model will make community-based research efforts more accessible, by releasing the weights of a highly performant 111 billion parameter model to researchers all over the world. This model is governed by a [CC-BY-NC](https://cohere.com/c4ai-cc-by-nc-license) License (Non-Commercial) with an acceptable use addendum, *and also requires adhering to [Cohere Lab's Acceptable Use Policy](https://docs.cohere.com/docs/c4ai-acceptable-use-policy)*. If you are interested in commercial use, please contact [Cohere’s Sales team](https://cohere.com/contact-sales). + +## **Try it now:** + +You can try Tiny Aya in our dedicated [Hugging Face Space](https://huggingface.co/spaces/CohereLabs/tiny-aya). + +## **Citation** + +``` +@misc{salamanca2026tinyayabridgingscale, + title={Tiny Aya: Bridging Scale and Multilingual Depth}, + author={Alejandro R. Salamanca and Diana Abagyan and Daniel D'souza and Ammar Khairi and David Mora and Saurabh Dash and Viraat Aryabumi and Sara Rajaee and Mehrnaz Mofakhami and Ananya Sahu and Thomas Euyang and Brittawnya Prince and Madeline Smith and Hangyu Lin and Acyr Locatelli and Sara Hooker and Tom Kocmi and Aidan Gomez and Ivan Zhang and Phil Blunsom and Nick Frosst and Joelle Pineau and Beyza Ermis and Ahmet Üstün and Julia Kreutzer and Marzieh Fadaee}, + year={2026}, + eprint={2603.11510}, + archivePrefix={arXiv}, + primaryClass={cs.CL}, + url={https://arxiv.org/abs/2603.11510}, +} +``` diff --git a/assets/TinyAyaPlot_D_Light.png b/assets/TinyAyaPlot_D_Light.png new file mode 100644 index 0000000..9e32709 Binary files /dev/null and b/assets/TinyAyaPlot_D_Light.png differ diff --git a/assets/TinyAya_Global.png b/assets/TinyAya_Global.png new file mode 100644 index 0000000..872dba6 --- /dev/null +++ b/assets/TinyAya_Global.png @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:68a845df08c9ebabaa2d74be188897854bfa0c3a9e2da9d6123f28559dfe50d2 +size 2997953 diff --git a/assets/TinyAya_PlotB_v7_lightmode.png b/assets/TinyAya_PlotB_v7_lightmode.png new file mode 100644 index 0000000..fef2781 Binary files /dev/null and b/assets/TinyAya_PlotB_v7_lightmode.png differ diff --git a/assets/tiny-aya-lowres-dotplot_lightmode.png b/assets/tiny-aya-lowres-dotplot_lightmode.png new file mode 100644 index 0000000..bea2eed Binary files /dev/null and b/assets/tiny-aya-lowres-dotplot_lightmode.png differ diff --git a/assets/tiny_aya_regional_heatmap_lightmode.png b/assets/tiny_aya_regional_heatmap_lightmode.png new file mode 100644 index 0000000..391721f Binary files /dev/null and b/assets/tiny_aya_regional_heatmap_lightmode.png differ diff --git a/config.json b/config.json new file mode 100644 index 0000000..30c1a79 --- /dev/null +++ b/config.json @@ -0,0 +1,79 @@ +{ + "_sliding_window_pattern": 4, + "architectures": [ + "Cohere2ForCausalLM" + ], + "attention_bias": false, + "attention_dropout": 0.0, + "bos_token_id": 2, + "cache_implementation": "hybrid", + "eos_token_id": 3, + "head_dim": 128, + "hidden_act": "silu", + "hidden_size": 2048, + "initializer_range": 0.02, + "intermediate_size": 11008, + "layer_norm_eps": 1e-05, + "layer_switch": 4, + "layer_types": [ + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention" + ], + "logit_scale": 1.0, + "max_position_embeddings": 8192, + "model_type": "cohere2", + "num_attention_heads": 16, + "num_hidden_layers": 36, + "num_key_value_heads": 4, + "order_of_interleaved_layers": "local_attn_first", + "pad_token_id": 0, + "position_embedding_type": "rope_gptj", + "rope_scaling": null, + "rope_theta": 50000, + "rotary_pct": 1.0, + "sliding_window": 4096, + "sliding_window_pattern": 4, + "torch_dtype": "bfloat16", + "transformers_version": "4.51.3", + "use_cache": true, + "use_embedding_sharing": true, + "use_gated_activation": true, + "use_parallel_block": true, + "use_parallel_embedding": false, + "use_qk_norm": false, + "vocab_size": 262144 +} diff --git a/generation_config.json b/generation_config.json new file mode 100644 index 0000000..a5c3b97 --- /dev/null +++ b/generation_config.json @@ -0,0 +1,7 @@ +{ + "_from_model_config": true, + "bos_token_id": 2, + "eos_token_id": 3, + "pad_token_id": 0, + "transformers_version": "4.51.3" +} diff --git a/model-00001-of-00002.safetensors b/model-00001-of-00002.safetensors new file mode 100644 index 0000000..0d5e8d8 --- /dev/null +++ b/model-00001-of-00002.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0590022d0efe06f12e08e1e94302f069dff404341b426f7435c6303725635fb1 +size 4992396352 diff --git a/model-00002-of-00002.safetensors b/model-00002-of-00002.safetensors new file mode 100644 index 0000000..107c2d7 --- /dev/null +++ b/model-00002-of-00002.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2ac3809931d113785058809f27775620e0b8aec6fbc7e163c9c09c08b5f90dab +size 1706092176 diff --git a/model.safetensors.index.json b/model.safetensors.index.json new file mode 100644 index 0000000..80c31f6 --- /dev/null +++ b/model.safetensors.index.json @@ -0,0 +1,297 @@ +{ + "metadata": { + "total_size": 6698455040 + }, + "weight_map": { + "model.embed_tokens.weight": "model-00001-of-00002.safetensors", + "model.layers.0.input_layernorm.weight": "model-00001-of-00002.safetensors", + "model.layers.0.mlp.down_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.0.mlp.gate_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.0.mlp.up_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.0.self_attn.k_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.0.self_attn.o_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.0.self_attn.q_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.0.self_attn.v_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.1.input_layernorm.weight": "model-00001-of-00002.safetensors", + "model.layers.1.mlp.down_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.1.mlp.gate_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.1.mlp.up_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.1.self_attn.k_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.1.self_attn.o_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.1.self_attn.q_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.1.self_attn.v_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.10.input_layernorm.weight": "model-00001-of-00002.safetensors", + "model.layers.10.mlp.down_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.10.mlp.gate_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.10.mlp.up_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.10.self_attn.k_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.10.self_attn.o_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.10.self_attn.q_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.10.self_attn.v_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.11.input_layernorm.weight": "model-00001-of-00002.safetensors", + "model.layers.11.mlp.down_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.11.mlp.gate_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.11.mlp.up_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.11.self_attn.k_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.11.self_attn.o_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.11.self_attn.q_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.11.self_attn.v_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.12.input_layernorm.weight": "model-00001-of-00002.safetensors", + "model.layers.12.mlp.down_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.12.mlp.gate_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.12.mlp.up_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.12.self_attn.k_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.12.self_attn.o_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.12.self_attn.q_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.12.self_attn.v_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.13.input_layernorm.weight": "model-00001-of-00002.safetensors", + "model.layers.13.mlp.down_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.13.mlp.gate_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.13.mlp.up_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.13.self_attn.k_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.13.self_attn.o_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.13.self_attn.q_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.13.self_attn.v_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.14.input_layernorm.weight": "model-00001-of-00002.safetensors", + "model.layers.14.mlp.down_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.14.mlp.gate_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.14.mlp.up_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.14.self_attn.k_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.14.self_attn.o_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.14.self_attn.q_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.14.self_attn.v_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.15.input_layernorm.weight": "model-00001-of-00002.safetensors", + "model.layers.15.mlp.down_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.15.mlp.gate_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.15.mlp.up_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.15.self_attn.k_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.15.self_attn.o_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.15.self_attn.q_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.15.self_attn.v_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.16.input_layernorm.weight": "model-00001-of-00002.safetensors", + "model.layers.16.mlp.down_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.16.mlp.gate_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.16.mlp.up_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.16.self_attn.k_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.16.self_attn.o_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.16.self_attn.q_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.16.self_attn.v_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.17.input_layernorm.weight": "model-00001-of-00002.safetensors", + "model.layers.17.mlp.down_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.17.mlp.gate_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.17.mlp.up_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.17.self_attn.k_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.17.self_attn.o_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.17.self_attn.q_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.17.self_attn.v_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.18.input_layernorm.weight": "model-00001-of-00002.safetensors", + "model.layers.18.mlp.down_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.18.mlp.gate_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.18.mlp.up_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.18.self_attn.k_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.18.self_attn.o_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.18.self_attn.q_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.18.self_attn.v_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.19.input_layernorm.weight": "model-00001-of-00002.safetensors", + "model.layers.19.mlp.down_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.19.mlp.gate_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.19.mlp.up_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.19.self_attn.k_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.19.self_attn.o_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.19.self_attn.q_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.19.self_attn.v_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.2.input_layernorm.weight": "model-00001-of-00002.safetensors", + "model.layers.2.mlp.down_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.2.mlp.gate_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.2.mlp.up_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.2.self_attn.k_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.2.self_attn.o_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.2.self_attn.q_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.2.self_attn.v_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.20.input_layernorm.weight": "model-00001-of-00002.safetensors", + "model.layers.20.mlp.down_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.20.mlp.gate_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.20.mlp.up_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.20.self_attn.k_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.20.self_attn.o_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.20.self_attn.q_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.20.self_attn.v_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.21.input_layernorm.weight": "model-00001-of-00002.safetensors", + "model.layers.21.mlp.down_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.21.mlp.gate_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.21.mlp.up_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.21.self_attn.k_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.21.self_attn.o_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.21.self_attn.q_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.21.self_attn.v_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.22.input_layernorm.weight": "model-00001-of-00002.safetensors", + "model.layers.22.mlp.down_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.22.mlp.gate_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.22.mlp.up_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.22.self_attn.k_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.22.self_attn.o_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.22.self_attn.q_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.22.self_attn.v_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.23.input_layernorm.weight": "model-00001-of-00002.safetensors", + "model.layers.23.mlp.down_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.23.mlp.gate_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.23.mlp.up_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.23.self_attn.k_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.23.self_attn.o_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.23.self_attn.q_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.23.self_attn.v_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.24.input_layernorm.weight": "model-00001-of-00002.safetensors", + "model.layers.24.mlp.down_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.24.mlp.gate_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.24.mlp.up_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.24.self_attn.k_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.24.self_attn.o_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.24.self_attn.q_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.24.self_attn.v_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.25.input_layernorm.weight": "model-00002-of-00002.safetensors", + "model.layers.25.mlp.down_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.25.mlp.gate_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.25.mlp.up_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.25.self_attn.k_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.25.self_attn.o_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.25.self_attn.q_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.25.self_attn.v_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.26.input_layernorm.weight": "model-00002-of-00002.safetensors", + "model.layers.26.mlp.down_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.26.mlp.gate_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.26.mlp.up_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.26.self_attn.k_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.26.self_attn.o_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.26.self_attn.q_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.26.self_attn.v_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.27.input_layernorm.weight": "model-00002-of-00002.safetensors", + "model.layers.27.mlp.down_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.27.mlp.gate_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.27.mlp.up_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.27.self_attn.k_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.27.self_attn.o_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.27.self_attn.q_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.27.self_attn.v_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.28.input_layernorm.weight": "model-00002-of-00002.safetensors", + "model.layers.28.mlp.down_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.28.mlp.gate_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.28.mlp.up_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.28.self_attn.k_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.28.self_attn.o_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.28.self_attn.q_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.28.self_attn.v_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.29.input_layernorm.weight": "model-00002-of-00002.safetensors", + "model.layers.29.mlp.down_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.29.mlp.gate_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.29.mlp.up_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.29.self_attn.k_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.29.self_attn.o_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.29.self_attn.q_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.29.self_attn.v_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.3.input_layernorm.weight": "model-00001-of-00002.safetensors", + "model.layers.3.mlp.down_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.3.mlp.gate_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.3.mlp.up_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.3.self_attn.k_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.3.self_attn.o_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.3.self_attn.q_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.3.self_attn.v_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.30.input_layernorm.weight": "model-00002-of-00002.safetensors", + "model.layers.30.mlp.down_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.30.mlp.gate_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.30.mlp.up_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.30.self_attn.k_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.30.self_attn.o_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.30.self_attn.q_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.30.self_attn.v_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.31.input_layernorm.weight": "model-00002-of-00002.safetensors", + "model.layers.31.mlp.down_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.31.mlp.gate_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.31.mlp.up_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.31.self_attn.k_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.31.self_attn.o_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.31.self_attn.q_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.31.self_attn.v_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.32.input_layernorm.weight": "model-00002-of-00002.safetensors", + "model.layers.32.mlp.down_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.32.mlp.gate_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.32.mlp.up_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.32.self_attn.k_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.32.self_attn.o_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.32.self_attn.q_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.32.self_attn.v_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.33.input_layernorm.weight": "model-00002-of-00002.safetensors", + "model.layers.33.mlp.down_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.33.mlp.gate_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.33.mlp.up_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.33.self_attn.k_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.33.self_attn.o_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.33.self_attn.q_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.33.self_attn.v_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.34.input_layernorm.weight": "model-00002-of-00002.safetensors", + "model.layers.34.mlp.down_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.34.mlp.gate_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.34.mlp.up_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.34.self_attn.k_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.34.self_attn.o_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.34.self_attn.q_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.34.self_attn.v_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.35.input_layernorm.weight": "model-00002-of-00002.safetensors", + "model.layers.35.mlp.down_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.35.mlp.gate_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.35.mlp.up_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.35.self_attn.k_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.35.self_attn.o_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.35.self_attn.q_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.35.self_attn.v_proj.weight": "model-00002-of-00002.safetensors", + "model.layers.4.input_layernorm.weight": "model-00001-of-00002.safetensors", + "model.layers.4.mlp.down_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.4.mlp.gate_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.4.mlp.up_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.4.self_attn.k_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.4.self_attn.o_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.4.self_attn.q_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.4.self_attn.v_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.5.input_layernorm.weight": "model-00001-of-00002.safetensors", + "model.layers.5.mlp.down_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.5.mlp.gate_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.5.mlp.up_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.5.self_attn.k_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.5.self_attn.o_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.5.self_attn.q_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.5.self_attn.v_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.6.input_layernorm.weight": "model-00001-of-00002.safetensors", + "model.layers.6.mlp.down_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.6.mlp.gate_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.6.mlp.up_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.6.self_attn.k_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.6.self_attn.o_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.6.self_attn.q_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.6.self_attn.v_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.7.input_layernorm.weight": "model-00001-of-00002.safetensors", + "model.layers.7.mlp.down_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.7.mlp.gate_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.7.mlp.up_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.7.self_attn.k_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.7.self_attn.o_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.7.self_attn.q_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.7.self_attn.v_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.8.input_layernorm.weight": "model-00001-of-00002.safetensors", + "model.layers.8.mlp.down_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.8.mlp.gate_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.8.mlp.up_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.8.self_attn.k_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.8.self_attn.o_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.8.self_attn.q_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.8.self_attn.v_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.9.input_layernorm.weight": "model-00001-of-00002.safetensors", + "model.layers.9.mlp.down_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.9.mlp.gate_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.9.mlp.up_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.9.self_attn.k_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.9.self_attn.o_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.9.self_attn.q_proj.weight": "model-00001-of-00002.safetensors", + "model.layers.9.self_attn.v_proj.weight": "model-00001-of-00002.safetensors", + "model.norm.weight": "model-00002-of-00002.safetensors" + } +} diff --git a/prompts/default.txt b/prompts/default.txt new file mode 100644 index 0000000..d0754cb --- /dev/null +++ b/prompts/default.txt @@ -0,0 +1,17 @@ +CONTEXT: +{context} + +TASK TYPE: `{task_type}` + +QUERY: +{query} + +Read the QUERY instructions carefully — this task type may be unfamiliar. Learn only from CONTEXT, then produce answers in exactly the form the QUERY asks for. + +How to solve: +1. Deduce the linguistic rules from the CONTEXT examples only. +2. Apply those rules to every item in QUERY, in order. +3. Match the answer unit and format shown by the instructions and CONTEXT (words, letters, digits, transcriptions, etc.). +4. Put only the bare required answer on each answer line — no glosses, numbering, or commentary. + +Draft, then verify completeness and format. Finally write a line that says exactly `FINAL ANSWERS:` and, below it, one bare answer per QUERY item in QUERY order — no numbering, no quotes, no extra text. diff --git a/prompts/fill_blanks.txt b/prompts/fill_blanks.txt new file mode 100644 index 0000000..f2b74d6 --- /dev/null +++ b/prompts/fill_blanks.txt @@ -0,0 +1,17 @@ +CONTEXT: +{context} + +TASK TYPE: `fill_blanks` + +QUERY: +{query} + +This is a fill-in-the-blanks task. Missing pieces may be whole words, morphemes, or phonetic segments — match the kind of unit CONTEXT uses in the corresponding positions. + +How to solve: +1. Study complete CONTEXT examples to recover the paradigm / pattern. +2. Identify exactly what is missing in each QUERY blank (not what could be invented). +3. Fill each blank with only the missing form, in the same style as CONTEXT (including transcription brackets if CONTEXT uses them). +4. Return one filled form per blank, in the order blanks / items appear in QUERY. + +Draft, then verify every blank has an answer. Finally write a line that says exactly `FINAL ANSWERS:` and, below it, one bare filled form per blank/item — no numbering, no quotes, no extra text. diff --git a/prompts/match_letters.txt b/prompts/match_letters.txt new file mode 100644 index 0000000..48b5c87 --- /dev/null +++ b/prompts/match_letters.txt @@ -0,0 +1,17 @@ +CONTEXT: +{context} + +TASK TYPE: `match_letters` + +QUERY: +{query} + +This is a matching / multiple-choice letter task. Each QUERY item asks you to choose among labeled options (A, B, C, ...). + +How to solve: +1. Extract the mapping or rule set from CONTEXT. +2. For each numbered QUERY item, evaluate the options against that rule set. +3. Return only the chosen option letter (A, B, C, ...), uppercase. +4. Do not repeat the option text, explanations, or punctuation around the letter on the answer lines. + +Draft, then check you have one letter per QUERY item. Finally write a line that says exactly `FINAL ANSWERS:` and, below it, one bare letter per item in QUERY order — no numbering, no quotes, no extra text. diff --git a/prompts/num_to_text.txt b/prompts/num_to_text.txt new file mode 100644 index 0000000..ee9d787 --- /dev/null +++ b/prompts/num_to_text.txt @@ -0,0 +1,17 @@ +CONTEXT: +{context} + +TASK TYPE: `num_to_text` + +QUERY: +{query} + +This is a number-to-text task. CONTEXT shows how digit values are written as number words (or expressions) in the target language; convert each QUERY number into that written form. + +How to solve: +1. Infer the number-word construction rules from CONTEXT (bases, multipliers, conjunctions, morphology). +2. Apply them to each QUERY number. +3. Return only the number written out in words / forms as in CONTEXT's language style. +4. Do not output digits on the answer lines and do not add English glosses or commentary. + +Draft, then verify each QUERY item has a written numeral form. Finally write a line that says exactly `FINAL ANSWERS:` and, below it, one bare written form per item in QUERY order — no numbering, no quotes, no extra text. diff --git a/prompts/text_to_num.txt b/prompts/text_to_num.txt new file mode 100644 index 0000000..c6a7bcf --- /dev/null +++ b/prompts/text_to_num.txt @@ -0,0 +1,17 @@ +CONTEXT: +{context} + +TASK TYPE: `text_to_num` + +QUERY: +{query} + +This is a text-to-number task. CONTEXT shows how number words (or number expressions) map to values; convert each QUERY expression into digits. + +How to solve: +1. Infer the number system from CONTEXT (bases, place values, multipliers, word order). +2. Apply it to each QUERY expression. +3. Return only the numeric value in ordinary digits (e.g. 42), with no units or words. +4. Do not spell numbers out and do not add commentary on the answer lines. + +Draft, then verify each QUERY item has a digit answer. Finally write a line that says exactly `FINAL ANSWERS:` and, below it, one bare number per item in QUERY order — no numbering, no quotes, no extra text. diff --git a/prompts/translation.txt b/prompts/translation.txt new file mode 100644 index 0000000..956a677 --- /dev/null +++ b/prompts/translation.txt @@ -0,0 +1,17 @@ +CONTEXT: +{context} + +TASK TYPE: `translation` + +QUERY: +{query} + +This is a translation task. Use only the CONTEXT examples to learn how forms map between languages (or between orthography and meaning), then translate each QUERY item. + +How to solve: +1. Align CONTEXT pairs and find systematic correspondences (roots, affixes, word order, agreement). +2. Apply those rules to each QUERY item in order. +3. Output only the translated form itself — in the language the QUERY asks for. +4. Do not add glosses, English explanations, punctuation wrappers, or commentary on the answer lines. + +Draft, then check completeness and format. Finally write a line that says exactly `FINAL ANSWERS:` and, below it, one bare translation per QUERY item, in QUERY order — no numbering, no quotes, no extra text. diff --git a/script.py b/script.py new file mode 100644 index 0000000..2dfed7b --- /dev/null +++ b/script.py @@ -0,0 +1,290 @@ +import os +import subprocess +import sys + + +def _install_bundled_deps() -> None: + """Install transformers from bundled wheels (eval sandbox has no PyPI access).""" + wheels_dir = os.path.join(os.path.dirname(os.path.abspath(__file__)), "wheels") + if not os.path.isdir(wheels_dir): + return + subprocess.run( + [ + sys.executable, + "-m", + "pip", + "install", + "-q", + "--no-index", + f"--find-links={wheels_dir}", + "transformers==4.56.2", + ], + check=True, + ) + + +_install_bundled_deps() + +import re +import csv +import json +import shutil +import tempfile +import torch +from transformers import AutoTokenizer, AutoModelForCausalLM + +# The repo is the working directory at run time, and there is no network. +os.environ["HF_HUB_OFFLINE"] = "1" +os.environ["TRANSFORMERS_OFFLINE"] = "1" +MODEL_ID = "." +MAX_NEW_TOKENS = 2000 +TEMPERATURE = 0.8 +TOP_P = 0.95 +MAX_ATTEMPTS = 5 +SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__)) +PROMPTS_DIR = os.path.join(SCRIPT_DIR, "prompts") + + +def load_tokenizer(model_id: str = "."): + """Load tokenizer, converting tokenizer.json for older tokenizers if needed.""" + tokenizer_path = os.path.join(model_id, "tokenizer.json") + with open(tokenizer_path, encoding="utf-8") as handle: + data = json.load(handle) + + merges = data.get("model", {}).get("merges", []) + if not merges or not isinstance(merges[0], list): + return AutoTokenizer.from_pretrained(model_id) + + # Older tokenizers expect merge pairs as "a b" strings, not ["a", "b"] lists. + data["model"]["merges"] = [" ".join(piece) for piece in merges] + tmpdir = tempfile.mkdtemp() + for name in ("tokenizer_config.json", "special_tokens_map.json"): + src = os.path.join(model_id, name) + if os.path.isfile(src): + shutil.copy(src, tmpdir) + with open(os.path.join(tmpdir, "tokenizer.json"), "w", encoding="utf-8") as handle: + json.dump(data, handle) + return AutoTokenizer.from_pretrained(tmpdir) + + +# Short durable contract only — task-specific procedure is loaded into the user turn. +SYSTEM = ( + "You solve International Linguistics Olympiad (IOL) problems using only the " + "CONTEXT and QUERY you are given. Do not rely on prior knowledge of the language.\n" + "Always end with a line that says exactly `FINAL ANSWERS:`, then one bare answer " + "per line for each QUERY item, in order — no numbering, quotes, or extra text." +) + +KNOWN_TASK_TYPES = ( + "translation", + "fill_blanks", + "match_letters", + "text_to_num", + "num_to_text", +) + + +def _prompt_path(task_type: str) -> str: + """Resolve prompts/.txt, falling back to prompts/default.txt.""" + # Keep basename-only to avoid path traversal if task_type is ever untrusted. + safe = os.path.basename(task_type.strip()) + candidate = os.path.join(PROMPTS_DIR, f"{safe}.txt") + if safe in KNOWN_TASK_TYPES and os.path.isfile(candidate): + return candidate + default = os.path.join(PROMPTS_DIR, "default.txt") + if os.path.isfile(default): + return default + raise FileNotFoundError( + f"No prompt template for task_type={task_type!r} under {PROMPTS_DIR}" + ) + + +def load_user_prompt_template(task_type: str) -> str: + with open(_prompt_path(task_type), encoding="utf-8") as handle: + return handle.read() + + +def build_user_prompt(context: str, task_type: str, query: str) -> str: + """Load the task-type-specific user template and fill in this example.""" + template = load_user_prompt_template(task_type) + return template.format( + context=context.strip(), + query=query.strip(), + task_type=task_type.strip(), + ) + + +# Prefer a dedicated header line; also allow same-line answers after the colon. +FINAL_ANSWERS_LINE_RE = re.compile( + r"(?im)^[^\w\n]*final answers?[^\w\n]*:?[ \t]*(?=\n|$)|" + r"(?im)^[^\w\n]*final answers?\s*:\s*" +) +FINAL_ANSWERS_INLINE_RE = re.compile( + r"(?is)\bfinal answers?\s*:\s*" +) + + +def extract_raw_final(text: str) -> str: + """Return text after the last final-answers marker, or '' if none found.""" + line_matches = list(FINAL_ANSWERS_LINE_RE.finditer(text)) + if line_matches: + return text[line_matches[-1].end() :] + + inline_matches = list(FINAL_ANSWERS_INLINE_RE.finditer(text)) + if inline_matches: + return text[inline_matches[-1].end() :] + + return "" + + +def expected_answer_count(query: str, task_type: str) -> int: + if task_type == "match_letters": + numbered = re.findall(r"^\s*\d+\.", query, re.MULTILINE) + return len(numbered) or 1 + + if "blanks" in query.lower(): + range_match = re.search(r"\((\d+)-(\d+)\)", query) + if range_match: + return int(range_match.group(2)) - int(range_match.group(1)) + 1 + return len(re.findall(r"\(\d+\)", query)) or 1 + + numbered = re.findall(r"^\s*\d+[.)]", query, re.MULTILINE) + return len(numbered) or 1 + + +def split_single_line_answer(text: str, expected: int, task_type: str) -> list[str]: + text = text.strip() + if expected <= 1: + return [text] + + def try_split(pattern: str) -> list[str] | None: + parts = [part.strip() for part in re.split(pattern, text) if part.strip()] + return parts if len(parts) == expected else None + + if task_type == "match_letters": + for pattern in (r"\s+", r",\s*", r";\s*"): + if result := try_split(pattern): + return result + letters = re.findall(r"[A-Za-z]", text) + if len(letters) == expected: + return [letter.upper() for letter in letters] + return [text] + + if task_type in ("text_to_num", "num_to_text"): + for pattern in (r",\s*", r";\s*", r"\s+"): + if result := try_split(pattern): + return result + return [text] + + for pattern in (r";\s*", r",\s*"): + if result := try_split(pattern): + return result + return [text] + + +def parse_answer_lines(text_after_marker: str, query: str, task_type: str) -> list[str]: + """Parse cleaned answer lines from the raw final-answers section.""" + answers = [] + for line in text_after_marker.splitlines(): + stripped_line = line.strip("`").strip() + if stripped_line == "": + continue + + match_numbered_prefix = re.match(r"^\s*\d+[.)]\s+(.*)", stripped_line) + if match_numbered_prefix: + cleaned_line = match_numbered_prefix.group(1).strip() + else: + cleaned_line = stripped_line + + cleaned_line = re.sub(r"\*\*", "", cleaned_line).strip() + + if task_type == "match_letters": + parts = [ + part.strip("().[]") + for part in re.split(r"[\s,;]+", cleaned_line) + if part.strip() + ] + if not ( + len(parts) > 1 + and all(re.fullmatch(r"[A-Za-z]", part) for part in parts) + ): + match_letter_word = re.match( + r"^\s*(?:\(([A-Za-z])\)|\[([A-Za-z])\]|([A-Za-z]))\.?:?\s*(.*)$", + cleaned_line, + ) + if match_letter_word: + letter = ( + match_letter_word.group(1) + or match_letter_word.group(2) + or match_letter_word.group(3) + ) + cleaned_line = letter.upper() + + if cleaned_line: + answers.append(cleaned_line) + + expected = expected_answer_count(query, task_type) + if len(answers) == 1 and expected > 1: + answers = split_single_line_answer(answers[0], expected, task_type) + return answers + + +def postprocess_answer(text, query, task_type): + """Keep only the content after the last 'FINAL ANSWERS' marker.""" + text_after_marker = extract_raw_final(text) + if not text_after_marker.strip(): + return [] + return parse_answer_lines(text_after_marker, query, task_type) + + +tok = load_tokenizer(MODEL_ID) +model = AutoModelForCausalLM.from_pretrained( + MODEL_ID, torch_dtype=torch.float16, device_map="auto" +).eval() + +with open("/tmp/data/test.csv", encoding="utf-8", newline="") as f: + test_rows = list(csv.DictReader(f)) + +outputs_queries_types = [] +for r in test_rows: + user_prompt = build_user_prompt(r["context"], r["task_type"], r["query"]) + print(f"id={r['id']} task_type={r['task_type']} template={os.path.basename(_prompt_path(r['task_type']))}", flush=True) + messages = [ + {"role": "system", "content": SYSTEM}, + {"role": "user", "content": user_prompt}, + ] + ids = tok.apply_chat_template( + messages, add_generation_prompt=True, return_tensors="pt", + ).to(model.device) + + done = False + attempts = 0 + text = "" + while not done and attempts < MAX_ATTEMPTS: + with torch.no_grad(): + out = model.generate( + ids, + max_new_tokens=MAX_NEW_TOKENS, + do_sample=True, + temperature=TEMPERATURE, + top_p=TOP_P, + ) + text = tok.decode(out[0][ids.shape[-1] :], skip_special_tokens=True).strip() + attempts += 1 + if extract_raw_final(text).strip(): + done = True + else: + print(f"TRYING AGAIN...attempts #{attempts}/{MAX_ATTEMPTS}", flush=True) + + outputs_queries_types.append((text, r["id"], r["query"], r["task_type"])) + print(f"{len(outputs_queries_types)}/{len(test_rows)} done", flush=True) + +rows = [] +for answer, row_id, query, task_type in outputs_queries_types: + answers = postprocess_answer(answer, query, task_type) + rows.append({"id": row_id, "pred": json.dumps(answers, ensure_ascii=False)}) +with open("submission.csv", "w", encoding="utf-8", newline="") as f: + writer = csv.DictWriter(f, fieldnames=["id", "pred"]) + writer.writeheader() + writer.writerows(rows) +print("wrote submission.csv", flush=True) diff --git a/signatures/tiny-aya-global.sig b/signatures/tiny-aya-global.sig new file mode 100644 index 0000000..6d176fe --- /dev/null +++ b/signatures/tiny-aya-global.sig @@ -0,0 +1 @@ +{"mediaType":"application/vnd.dev.sigstore.bundle.v0.3+json","verificationMaterial":{"certificate":{"rawBytes":"MIIHADCCBoagAwIBAgIUZIcMI+ZPc0S31km3K68QbHVvPhIwCgYIKoZIzj0EAwMwNzEVMBMGA1UEChMMc2lnc3RvcmUuZGV2MR4wHAYDVQQDExVzaWdzdG9yZS1pbnRlcm1lZGlhdGUwHhcNMjYwMjI0MDgwNzMyWhcNMjYwMjI0MDgxNzMyWjAAMFkwEwYHKoZIzj0CAQYIKoZIzj0DAQcDQgAEIhFKfYXkqzvHEBgdd59ZD4ZtO/9E5ONI5jlD4rk0gM36EVCYIE6qVcCaLyRy+kVET903VeHi0VWxAl7IdrSoe6OCBaUwggWhMA4GA1UdDwEB/wQEAwIHgDATBgNVHSUEDDAKBggrBgEFBQcDAzAdBgNVHQ4EFgQUi2LIjlId5nrvt2fInix+lfAW7oUwHwYDVR0jBBgwFoAU39Ppz1YkEZb5qNjpKFWixi4YZD8waQYDVR0RAQH/BF8wXYZbaHR0cHM6Ly9naXRodWIuY29tL2NvaGVyZS1haS9tb2RlbC1zaWduaW5nLy5naXRodWIvd29ya2Zsb3dzL3NpZ24tbW9kZWwueW1sQHJlZnMvaGVhZHMvbWFpbjA5BgorBgEEAYO/MAEBBCtodHRwczovL3Rva2VuLmFjdGlvbnMuZ2l0aHVidXNlcmNvbnRlbnQuY29tMB8GCisGAQQBg78wAQIEEXdvcmtmbG93X2Rpc3BhdGNoMDYGCisGAQQBg78wAQMEKDRlNDg2OTI3MDM5MDViOTJiMDMyYzY3NDNhZWE4YmFiOWYyNDU0M2IwJgYKKwYBBAGDvzABBAQYU2lnbiBNb2RlbCB3aXRoIFNpZ3N0b3JlMCUGCisGAQQBg78wAQUEF2NvaGVyZS1haS9tb2RlbC1zaWduaW5nMB0GCisGAQQBg78wAQYED3JlZnMvaGVhZHMvbWFpbjA7BgorBgEEAYO/MAEIBC0MK2h0dHBzOi8vdG9rZW4uYWN0aW9ucy5naXRodWJ1c2VyY29udGVudC5jb20wawYKKwYBBAGDvzABCQRdDFtodHRwczovL2dpdGh1Yi5jb20vY29oZXJlLWFpL21vZGVsLXNpZ25pbmcvLmdpdGh1Yi93b3JrZmxvd3Mvc2lnbi1tb2RlbC55bWxAcmVmcy9oZWFkcy9tYWluMDgGCisGAQQBg78wAQoEKgwoNGU0ODY5MjcwMzkwNWI5MmIwMzJjNjc0M2FlYThiYWI5ZjI0NTQzYjAdBgorBgEEAYO/MAELBA8MDWdpdGh1Yi1ob3N0ZWQwOgYKKwYBBAGDvzABDAQsDCpodHRwczovL2dpdGh1Yi5jb20vY29oZXJlLWFpL21vZGVsLXNpZ25pbmcwOAYKKwYBBAGDvzABDQQqDCg0ZTQ4NjkyNzAzOTA1YjkyYjAzMmM2NzQzYWVhOGJhYjlmMjQ1NDNiMB8GCisGAQQBg78wAQ4EEQwPcmVmcy9oZWFkcy9tYWluMBoGCisGAQQBg78wAQ8EDAwKMTA2NzY3NTI1MDAsBgorBgEEAYO/MAEQBB4MHGh0dHBzOi8vZ2l0aHViLmNvbS9jb2hlcmUtYWkwGAYKKwYBBAGDvzABEQQKDAg1NDg1MDkyMzBrBgorBgEEAYO/MAESBF0MW2h0dHBzOi8vZ2l0aHViLmNvbS9jb2hlcmUtYWkvbW9kZWwtc2lnbmluZy8uZ2l0aHViL3dvcmtmbG93cy9zaWduLW1vZGVsLnltbEByZWZzL2hlYWRzL21haW4wOAYKKwYBBAGDvzABEwQqDCg0ZTQ4NjkyNzAzOTA1YjkyYjAzMmM2NzQzYWVhOGJhYjlmMjQ1NDNiMCEGCisGAQQBg78wARQEEwwRd29ya2Zsb3dfZGlzcGF0Y2gwXgYKKwYBBAGDvzABFQRQDE5odHRwczovL2dpdGh1Yi5jb20vY29oZXJlLWFpL21vZGVsLXNpZ25pbmcvYWN0aW9ucy9ydW5zLzIyMzQyMDIyNzkwL2F0dGVtcHRzLzEwGAYKKwYBBAGDvzABFgQKDAhpbnRlcm5hbDCBigYKKwYBBAHWeQIEAgR8BHoAeAB2AN09MGrGxxEyYxkeHJlnNwKiSl643jyt/4eKcoAvKe6OAAABnI6waz8AAAQDAEcwRQIgZz2S1U/qPob3skjBvIJa2ozf4Nk76OSZnXyOx8DVAHcCIQC6FCfDyUqdxD2lWOA8e0SoVRmOwDuF/4unbBZJ0UXOiDAKBggqhkjOPQQDAwNoADBlAjEAtYt/GWkm3ie8EFdHAvsdkp6n6D8Os2rHlnzzbgOoonFThPTLGW9i66QpZ8PCnirvAjBnaR0RZ+mgo/7kP+Jlp8hsTXCix1lBNYhJf4p1gjq3MxCIDS67wb9mlAsWzbNUy3Q="},"tlogEntries":[{"logIndex":"984891074","logId":{"keyId":"wNI9atQGlz+VWfO6LRygH4QUfY/8W4RFwiT5i5WRgB0="},"kindVersion":{"kind":"dsse","version":"0.0.1"},"integratedTime":"1771920452","inclusionPromise":{"signedEntryTimestamp":"MEQCICZYzuxBGgOJkN4lWn8KYk+wOrWXkl/gZgKd4EFrvdt8AiARowV1LWr0onrXjfTSYwhcZH/7xPvzG+JGbRCKcpTPhg=="},"inclusionProof":{"logIndex":"862986812","rootHash":"jBYzcKqwYKakISvS3SXOnMTgZupPlwGRRRt6vP4rHFM=","treeSize":"862986813","hashes":["W6QNxcmPw5I8Cl9Ldwr5ThePClWRqV5Zy+HBfNj4tac=","DVw0kfFv8cc+A5sRrQW42kjXxiIMIMPowZ7Ci1Hhhns=","xnueeeofYa4NlBrr6DURb5+e2ORzxNfzcuY4xYj4Zfs=","MTqL26qZk4YvstZfnXqCYfyRMnH0GzVVGQj3hnLNISI=","54PsHy6hR6SuAGwR6Rdex71LVYrZl08BCCwp9knyoaA=","xXpTls0Hlan5ozWoWVaWHsh5zo7HNVLiwacLgfUgfZw=","FLeXvOaL+UkHG7v34RhFX1wmnQzVbQ8Ne/mwCJuKKMQ=","Bh7k8hWxwOrj+Un2vU0UhQU/2e46PDwZE5r+2bgc4yI=","BFnR7niej8x7wrDi7Bc6sExDUe0ZN0brOtSnvBTSsuE=","RHXDRAW0fmTpkqS38IgfdjG+/M0tCJadplFys+5MMl0=","Cf3Qsee7cmtATi56kFqvoGskpRx2bvx5qhyhiqobL0U=","fLAvE46NqCVV86EpB2pKkwJlFjjFk7ntX3lC+PiZuIo=","T4DqWD42hAtN+vX8jKCWqoC4meE4JekI9LxYGCcPy1M="],"checkpoint":{"envelope":"rekor.sigstore.dev - 1193050959916656506\n862986813\njBYzcKqwYKakISvS3SXOnMTgZupPlwGRRRt6vP4rHFM=\n\n— rekor.sigstore.dev wNI9ajBEAiAy28N508KgC9VB4/+WaTYPBMbgfU48KSVo0THKPG0YwgIgZ7TwD7cNWNXw0p7Kclat/YnEhynwxTMitLGil1qznNA=\n"}},"canonicalizedBody":"eyJhcGlWZXJzaW9uIjoiMC4wLjEiLCJraW5kIjoiZHNzZSIsInNwZWMiOnsiZW52ZWxvcGVIYXNoIjp7ImFsZ29yaXRobSI6InNoYTI1NiIsInZhbHVlIjoiOTQyOGYwMzBhMzE5ODZmM2I3NDllN2JkNTU4ODBmMGQ4ZDRjZTg5NmMzNzRiMWQ5Mjg4NjBiYmM0YzU2MDFkZSJ9LCJwYXlsb2FkSGFzaCI6eyJhbGdvcml0aG0iOiJzaGEyNTYiLCJ2YWx1ZSI6IjRmNzM3ZjlhYzIyMWM0ZTRjMDc0MDEyYzQyYTRmM2M4Y2I1ZDMyM2FmOTU1YTUxZDA0NGEzY2UwMTVkMGI3OGQifSwic2lnbmF0dXJlcyI6W3sic2lnbmF0dXJlIjoiTUVZQ0lRQ1BXSS80MXBOYXdNR2NHZHdCRkROV3EwMllmWlpYc1g3d0RCa1JySjRDQ1FJaEFLeElBYWZUbG1nQkh3T3NPTEd2KzVPc3RSUWx1Yk9vK3JXUERGZFZpVFJyIiwidmVyaWZpZXIiOiJMUzB0TFMxQ1JVZEpUaUJEUlZKVVNVWkpRMEZVUlMwdExTMHRDazFKU1VoQlJFTkRRbTloWjBGM1NVSkJaMGxWV2tsalRVa3JXbEJqTUZNek1XdHRNMHMyT0ZGaVNGWjJVR2hKZDBObldVbExiMXBKZW1vd1JVRjNUWGNLVG5wRlZrMUNUVWRCTVZWRlEyaE5UV015Ykc1ak0xSjJZMjFWZFZwSFZqSk5ValIzU0VGWlJGWlJVVVJGZUZaNllWZGtlbVJIT1hsYVV6RndZbTVTYkFwamJURnNXa2RzYUdSSFZYZElhR05PVFdwWmQwMXFTVEJOUkdkM1RucE5lVmRvWTA1TmFsbDNUV3BKTUUxRVozaE9lazE1VjJwQlFVMUdhM2RGZDFsSUNrdHZXa2w2YWpCRFFWRlpTVXR2V2tsNmFqQkVRVkZqUkZGblFVVkphRVpMWmxsWWEzRjZka2hGUW1ka1pEVTVXa1EwV25SUEx6bEZOVTlPU1RWcWJFUUtOSEpyTUdkTk16WkZWa05aU1VVMmNWWmpRMkZNZVZKNUsydFdSVlE1TUROV1pVaHBNRlpYZUVGc04wbGtjbE52WlRaUFEwSmhWWGRuWjFkb1RVRTBSd3BCTVZWa1JIZEZRaTkzVVVWQmQwbElaMFJCVkVKblRsWklVMVZGUkVSQlMwSm5aM0pDWjBWR1FsRmpSRUY2UVdSQ1owNVdTRkUwUlVablVWVnBNa3hKQ21wc1NXUTFibkoyZERKbVNXNXBlQ3RzWmtGWE4yOVZkMGgzV1VSV1VqQnFRa0puZDBadlFWVXpPVkJ3ZWpGWmEwVmFZalZ4VG1wd1MwWlhhWGhwTkZrS1drUTRkMkZSV1VSV1VqQlNRVkZJTDBKR09IZFlXVnBpWVVoU01HTklUVFpNZVRsdVlWaFNiMlJYU1hWWk1qbDBUREpPZG1GSFZubGFVekZvWVZNNWRBcGlNbEpzWWtNeGVtRlhaSFZoVnpWdVRIazFibUZZVW05a1YwbDJaREk1ZVdFeVduTmlNMlI2VEROT2NGb3lOSFJpVnpscldsZDNkV1ZYTVhOUlNFcHNDbHB1VFhaaFIxWm9Xa2hOZG1KWFJuQmlha0UxUW1kdmNrSm5SVVZCV1U4dlRVRkZRa0pEZEc5a1NGSjNZM3B2ZGt3elVuWmhNbFoxVEcxR2FtUkhiSFlLWW01TmRWb3liREJoU0ZacFpGaE9iR050VG5aaWJsSnNZbTVSZFZreU9YUk5RamhIUTJselIwRlJVVUpuTnpoM1FWRkpSVVZZWkhaamJYUnRZa2M1TXdwWU1sSndZek5DYUdSSFRtOU5SRmxIUTJselIwRlJVVUpuTnpoM1FWRk5SVXRFVW14T1JHY3lUMVJKTTAxRVRUVk5SRlpwVDFSS2FVMUVUWGxaZWxrekNrNUVUbWhhVjBVMFdXMUdhVTlYV1hsT1JGVXdUVEpKZDBwbldVdExkMWxDUWtGSFJIWjZRVUpDUVZGWlZUSnNibUpwUWs1aU1sSnNZa05DTTJGWVVtOEtTVVpPY0ZvelRqQmlNMHBzVFVOVlIwTnBjMGRCVVZGQ1p6YzRkMEZSVlVWR01rNTJZVWRXZVZwVE1XaGhVemwwWWpKU2JHSkRNWHBoVjJSMVlWYzFiZ3BOUWpCSFEybHpSMEZSVVVKbk56aDNRVkZaUlVRelNteGFiazEyWVVkV2FGcElUWFppVjBad1ltcEJOMEpuYjNKQ1owVkZRVmxQTDAxQlJVbENRekJOQ2tzeWFEQmtTRUo2VDJrNGRtUkhPWEphVnpSMVdWZE9NR0ZYT1hWamVUVnVZVmhTYjJSWFNqRmpNbFo1V1RJNWRXUkhWblZrUXpWcVlqSXdkMkYzV1VzS1MzZFpRa0pCUjBSMmVrRkNRMUZTWkVSR2RHOWtTRkozWTNwdmRrd3laSEJrUjJneFdXazFhbUl5TUhaWk1qbHZXbGhLYkV4WFJuQk1NakYyV2tkV2N3cE1XRTV3V2pJMWNHSnRZM1pNYldSd1pFZG9NVmxwT1ROaU0wcHlXbTE0ZG1RelRYWmpNbXh1WW1reGRHSXlVbXhpUXpVMVlsZDRRV050Vm0xamVUbHZDbHBYUm10amVUbDBXVmRzZFUxRVowZERhWE5IUVZGUlFtYzNPSGRCVVc5RlMyZDNiMDVIVlRCUFJGazFUV3BqZDAxNmEzZE9WMGsxVFcxSmQwMTZTbW9LVG1wak1FMHlSbXhaVkdocFdWZEpOVnBxU1RCT1ZGRjZXV3BCWkVKbmIzSkNaMFZGUVZsUEwwMUJSVXhDUVRoTlJGZGtjR1JIYURGWmFURnZZak5PTUFwYVYxRjNUMmRaUzB0M1dVSkNRVWRFZG5wQlFrUkJVWE5FUTNCdlpFaFNkMk42YjNaTU1tUndaRWRvTVZscE5XcGlNakIyV1RJNWIxcFlTbXhNVjBad0Nrd3lNWFphUjFaelRGaE9jRm95TlhCaWJXTjNUMEZaUzB0M1dVSkNRVWRFZG5wQlFrUlJVWEZFUTJjd1dsUlJORTVxYTNsT2VrRjZUMVJCTVZscWEza0tXV3BCZWsxdFRUSk9lbEY2V1ZkV2FFOUhTbWhaYW14dFRXcFJNVTVFVG1sTlFqaEhRMmx6UjBGUlVVSm5OemgzUVZFMFJVVlJkMUJqYlZadFkzazVid3BhVjBaclkzazVkRmxYYkhWTlFtOUhRMmx6UjBGUlVVSm5OemgzUVZFNFJVUkJkMHROVkVFeVRucFpNMDVVU1RGTlJFRnpRbWR2Y2tKblJVVkJXVTh2Q2sxQlJWRkNRalJOU0Vkb01HUklRbnBQYVRoMldqSnNNR0ZJVm1sTWJVNTJZbE01YW1JeWFHeGpiVlYwV1ZkcmQwZEJXVXRMZDFsQ1FrRkhSSFo2UVVJS1JWRlJTMFJCWnpGT1JHY3hUVVJyZVUxNlFuSkNaMjl5UW1kRlJVRlpUeTlOUVVWVFFrWXdUVmN5YURCa1NFSjZUMms0ZGxveWJEQmhTRlpwVEcxT2RncGlVemxxWWpKb2JHTnRWWFJaVjJ0MllsYzVhMXBYZDNSak1teHVZbTFzZFZwNU9IVmFNbXd3WVVoV2FVd3paSFpqYlhSdFlrYzVNMk41T1hwaFYyUjFDa3hYTVhaYVIxWnpURzVzZEdKRlFubGFWMXA2VERKb2JGbFhVbnBNTWpGb1lWYzBkMDlCV1V0TGQxbENRa0ZIUkhaNlFVSkZkMUZ4UkVObk1GcFVVVFFLVG1wcmVVNTZRWHBQVkVFeFdXcHJlVmxxUVhwTmJVMHlUbnBSZWxsWFZtaFBSMHBvV1dwc2JVMXFVVEZPUkU1cFRVTkZSME5wYzBkQlVWRkNaemM0ZHdwQlVsRkZSWGQzVW1ReU9YbGhNbHB6WWpOa1pscEhiSHBqUjBZd1dUSm5kMWhuV1V0TGQxbENRa0ZIUkhaNlFVSkdVVkpSUkVVMWIyUklVbmRqZW05MkNrd3laSEJrUjJneFdXazFhbUl5TUhaWk1qbHZXbGhLYkV4WFJuQk1NakYyV2tkV2MweFlUbkJhTWpWd1ltMWpkbGxYVGpCaFZ6bDFZM2s1ZVdSWE5Yb0tUSHBKZVUxNlVYbE5SRWw1VG5wcmQwd3lSakJrUjFaMFkwaFNla3g2UlhkSFFWbExTM2RaUWtKQlIwUjJla0ZDUm1kUlMwUkJhSEJpYmxKc1kyMDFhQXBpUkVOQ2FXZFpTMHQzV1VKQ1FVaFhaVkZKUlVGblVqaENTRzlCWlVGQ01rRk9NRGxOUjNKSGVIaEZlVmw0YTJWSVNteHVUbmRMYVZOc05qUXphbmwwQ2k4MFpVdGpiMEYyUzJVMlQwRkJRVUp1U1RaM1lYbzRRVUZCVVVSQlJXTjNVbEZKWjFwNk1sTXhWUzl4VUc5aU0zTnJha0oyU1VwaE1tOTZaalJPYXpjS05rOVRXbTVZZVU5NE9FUldRVWhqUTBsUlF6WkdRMlpFZVZWeFpIaEVNbXhYVDBFNFpUQlRiMVpTYlU5M1JIVkdMelIxYm1KQ1drb3dWVmhQYVVSQlN3cENaMmR4YUd0cVQxQlJVVVJCZDA1dlFVUkNiRUZxUlVGMFdYUXZSMWRyYlROcFpUaEZSbVJJUVhaelpHdHdObTQyUkRoUGN6SnlTR3h1ZW5waVowOXZDbTl1UmxSb1VGUk1SMWM1YVRZMlVYQmFPRkJEYm1seWRrRnFRbTVoVWpCU1dpdHRaMjh2TjJ0UUswcHNjRGhvYzFSWVEybDRNV3hDVGxsb1NtWTBjREVLWjJweE0wMTRRMGxFVXpZM2QySTViV3hCYzFkNllrNVZlVE5SUFFvdExTMHRMVVZPUkNCRFJWSlVTVVpKUTBGVVJTMHRMUzB0Q2c9PSJ9XX19"}],"timestampVerificationData":{"rfc3161Timestamps":[{"signedTimestamp":"MIIE6TADAgEAMIIE4AYJKoZIhvcNAQcCoIIE0TCCBM0CAQMxDTALBglghkgBZQMEAgEwgcIGCyqGSIb3DQEJEAEEoIGyBIGvMIGsAgEBBgkrBgEEAYO/MAIwMTANBglghkgBZQMEAgEFAAQgeEmOvGbauxTBbsODsQJosZUgR0IzlmN0mGgbAU85j84CFF0iF3Dj/cEIi7lY6/hO4b2/oMdgGA8yMDI2MDIyNDA4MDczMlowAwIBAQIJAJHhmg0l2aFNoDKkMDAuMRUwEwYDVQQKEwxzaWdzdG9yZS5kZXYxFTATBgNVBAMTDHNpZ3N0b3JlLXRzYaCCAhQwggIQMIIBlqADAgECAhQ6E1QvDJBh7rzBQy/Lio6LKiOLDDAKBggqhkjOPQQDAzA5MRUwEwYDVQQKEwxzaWdzdG9yZS5kZXYxIDAeBgNVBAMTF3NpZ3N0b3JlLXRzYS1zZWxmc2lnbmVkMB4XDTI1MDQwODA2NTk0M1oXDTM1MDQwNjA2NTk0M1owLjEVMBMGA1UEChMMc2lnc3RvcmUuZGV2MRUwEwYDVQQDEwxzaWdzdG9yZS10c2EwdjAQBgcqhkjOPQIBBgUrgQQAIgNiAATitrZnyEo2KDZP2QWMIBOgYbfSOTL5ZC/cHMv6Yq+HVIo1H9TC7Cx80KDiyvKhgB3wTqKyi9UDczhqg12b1AOLnRnydMTK+qB8M+1MjBci1+Jb8AV/VXu7CRuQCiPTHFyjajBoMA4GA1UdDwEB/wQEAwIHgDAdBgNVHQ4EFgQUif15Q4fP0GVGwwJGxyxzW3206wMwHwYDVR0jBBgwFoAUmOwB73+7Uf/UlR5vioiYUweJzr8wFgYDVR0lAQH/BAwwCgYIKwYBBQUHAwgwCgYIKoZIzj0EAwMDaAAwZQIwO2mxX/opo7SrIX9QyxfZpJRcpAV2gZOm1AZzR+2rVyy6Uc8Ybp2ybIw13ckH4bcRAjEA5qO8FyOkmYpvg2/7ZNqiPxRzn5vqKHoVcIIqtpKq6l7TvOqzAxxclN7VwTG8e++XMYIB2jCCAdYCAQEwUTA5MRUwEwYDVQQKEwxzaWdzdG9yZS5kZXYxIDAeBgNVBAMTF3NpZ3N0b3JlLXRzYS1zZWxmc2lnbmVkAhQ6E1QvDJBh7rzBQy/Lio6LKiOLDDALBglghkgBZQMEAgGggfwwGgYJKoZIhvcNAQkDMQ0GCyqGSIb3DQEJEAEEMBwGCSqGSIb3DQEJBTEPFw0yNjAyMjQwODA3MzJaMC8GCSqGSIb3DQEJBDEiBCDFqjtEqSb6F8JtBJ5sLlW8IWFPxcijaofIHmruxS7xnDCBjgYLKoZIhvcNAQkQAi8xfzB9MHsweQQghfknvAerYsrDtENWwQ78gbLGiD/aernm2HDZ0TrNBbcwVTA9pDswOTEVMBMGA1UEChMMc2lnc3RvcmUuZGV2MSAwHgYDVQQDExdzaWdzdG9yZS10c2Etc2VsZnNpZ25lZAIUOhNULwyQYe68wUMvy4qOiyojiwwwCgYIKoZIzj0EAwIEZjBkAjAFKpsyPcuFXKIsaa4AniZwP4UZvzXclzfmed5qkHJiZ7D4uK1HttS0UQGJBYQUUfUCMByMig8gDCQA1887Rm7aXkV8U8HI7essNNnFYSAX4MWUnKEUkI2zTMCbF0pEeMkXZA=="}]}},"dsseEnvelope":{"payload":"ewogICJfdHlwZSI6ICJodHRwczovL2luLXRvdG8uaW8vU3RhdGVtZW50L3YxIiwKICAic3ViamVjdCI6IFsKICAgIHsKICAgICAgIm5hbWUiOiAibW9kZWxfY2FjaGUiLAogICAgICAiZGlnZXN0IjogewogICAgICAgICJzaGEyNTYiOiAiNzBlYWEzM2IxNDVmY2EzZmU3MzAwYzNlOGFkNjUyY2NkNjMzNDc2MjVmYTZhZjMxMjkzY2I5N2I3OTlmZjk4YSIKICAgICAgfQogICAgfQogIF0sCiAgInByZWRpY2F0ZVR5cGUiOiAiaHR0cHM6Ly9tb2RlbF9zaWduaW5nL3NpZ25hdHVyZS92MS4wIiwKICAicHJlZGljYXRlIjogewogICAgInNlcmlhbGl6YXRpb24iOiB7CiAgICAgICJoYXNoX3R5cGUiOiAic2hhMjU2IiwKICAgICAgIm1ldGhvZCI6ICJmaWxlcyIsCiAgICAgICJhbGxvd19zeW1saW5rcyI6IGZhbHNlLAogICAgICAiaWdub3JlX3BhdGhzIjogWwogICAgICAgICIuZ2l0aHViIiwKICAgICAgICAiLmdpdGlnbm9yZSIsCiAgICAgICAgIi5naXRhdHRyaWJ1dGVzIiwKICAgICAgICAiLmdpdCIKICAgICAgXQogICAgfSwKICAgICJyZXNvdXJjZXMiOiBbCiAgICAgIHsKICAgICAgICAibmFtZSI6ICJjb25maWcuanNvbiIsCiAgICAgICAgImRpZ2VzdCI6ICJmMGVhZDI0ZTBhMTdhNTNmZmMzZjI1YWMzMjk4MjNhYThiOTUxNzYwOWRhYzE5ZDU1YzJhODRkOGUyYTUwZjA5IiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogImdlbmVyYXRpb25fY29uZmlnLmpzb24iLAogICAgICAgICJkaWdlc3QiOiAiMGVmYTY5NmUxZWJjMDQ4NGY4ZTE2ZmNkMTE0MWRiYTFlODk3YzQ2OTlmZjk3MTA1MmU0M2NlNWQ2ZDRiZjRkMCIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAibmFtZSI6ICJtb2RlbC0wMDAwMS1vZi0wMDAwMi5zYWZldGVuc29ycyIsCiAgICAgICAgImRpZ2VzdCI6ICIwNTkwMDIyZDBlZmUwNmYxMmUwOGUxZTk0MzAyZjA2OWRmZjQwNDM0MWI0MjZmNzQzNWM2MzAzNzI1NjM1ZmIxIiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogIm1vZGVsLTAwMDAyLW9mLTAwMDAyLnNhZmV0ZW5zb3JzIiwKICAgICAgICAiZGlnZXN0IjogIjJhYzM4MDk5MzFkMTEzNzg1MDU4ODA5ZjI3Nzc1NjIwZTBiOGFlYzZmYmM3ZTE2M2M5YzA5YzA4YjVmOTBkYWIiLAogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IgogICAgICB9LAogICAgICB7CiAgICAgICAgIm5hbWUiOiAibW9kZWwuc2FmZXRlbnNvcnMuaW5kZXguanNvbiIsCiAgICAgICAgImRpZ2VzdCI6ICI1ZTg3YmU3MTdmZjg4ZWQ3ZmQxZTAzYWExZGUxMzk5ZGFlZjJmMmQyZTNlZTUzOWMzYWQxYjBlZDlkMWFlNzRlIiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogInNwZWNpYWxfdG9rZW5zX21hcC5qc29uIiwKICAgICAgICAiZGlnZXN0IjogIjJiN2NlMzk2MzBmOGExODcwNTMzMWE5YjMyYzNiMjIzNmU5NmU3NDEwNzkxZjk4YTlmOGI2NDk0ZTNhYjJlYTYiLAogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IgogICAgICB9LAogICAgICB7CiAgICAgICAgIm5hbWUiOiAidG9rZW5pemVyLmpzb24iLAogICAgICAgICJkaWdlc3QiOiAiMjIyN2VhOWM1MmU4YWZiM2Y5OGJmZWQyNjc5MDA4YjI3NWYyNjY0ZGU2OWRmZGUxNzRiMzc0Mzg5ZWIwMjI1ZCIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAibmFtZSI6ICJ0b2tlbml6ZXJfY29uZmlnLmpzb24iLAogICAgICAgICJkaWdlc3QiOiAiOTg1NDFmOTdjNGM1OGVjYWMwNzQ2YTJiNjc3ZjczZmY4ZTY1YThjZWEwMDJhMzM3ODczZGUyMmY0ZDQ4MTRmOSIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiCiAgICAgIH0KICAgIF0KICB9Cn0=","payloadType":"application/vnd.in-toto+json","signatures":[{"sig":"MEYCIQCPWI/41pNawMGcGdwBFDNWq02YfZZXsX7wDBkRrJ4CCQIhAKxIAafTlmgBHwOsOLGv+5OstRQlubOo+rWPDFdViTRr"}]}} \ No newline at end of file diff --git a/signatures/verification-instructions.txt b/signatures/verification-instructions.txt new file mode 100644 index 0000000..973b1c0 --- /dev/null +++ b/signatures/verification-instructions.txt @@ -0,0 +1,40 @@ +==================================== +MODEL SIGNATURE VERIFICATION GUIDE +==================================== + +Model: CohereLabs/tiny-aya-global +Revision: main +Environment: PRODUCTION +Signed at: 2025-10-27T18:55:09Z +Workflow Run: https://github.com/cohere-ai/model-signing/actions/runs/22342022790 + +TRANSPARENCY LOG +---------------- +This signature is recorded in the Sigstore Rekor transparency log. + +Rekor Entry: https://search.sigstore.dev/?logIndex=984891074 +Log Index: 984891074 +Identity: https://github.com/cohere-ai/model-signing/.github/workflows/sign-model.yml@refs/heads/main + +VERIFICATION +------------ +To verify this signature locally: + +1. Install the model-signing package: + pip install model-signing + +2. Install huggingface_hub and download the model: + pip install huggingface_hub + huggingface-cli download CohereLabs/tiny-aya-global --revision main --local-dir ./model + +3. Verify the signature: + model_signing verify ./model \ + --signature tiny-aya-global.sig \ + --identity "https://github.com/cohere-ai/model-signing/.github/workflows/sign-model.yml@refs/heads/main" \ + --identity_provider "https://token.actions.githubusercontent.com" \ + --ignore_unsigned_files + +Note: This signature was created with selective file inclusion (*.safetensors,*.bin,*.json,*.txt,*.model,*.yaml,*.yml). + Use --ignore_unsigned_files to verify only the files that were signed. + +==================================== diff --git a/special_tokens_map.json b/special_tokens_map.json new file mode 100644 index 0000000..ca863ef --- /dev/null +++ b/special_tokens_map.json @@ -0,0 +1,30 @@ +{ + "bos_token": { + "content": "", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false + }, + "eos_token": { + "content": "", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false + }, + "pad_token": { + "content": "", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false + }, + "unk_token": { + "content": "", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false + } +} diff --git a/tokenizer.json b/tokenizer.json new file mode 100644 index 0000000..9fe05e3 --- /dev/null +++ b/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2227ea9c52e8afb3f98bfed2679008b275f2664de69dfde174b374389eb0225d +size 21376527 diff --git a/tokenizer_config.json b/tokenizer_config.json new file mode 100644 index 0000000..b6f2eb6 --- /dev/null +++ b/tokenizer_config.json @@ -0,0 +1,214 @@ +{ + "add_bos_token": true, + "add_eos_token": false, + "add_prefix_space": false, + "added_tokens_decoder": { + "0": { + "content": "", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "1": { + "content": "", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "2": { + "content": "", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "3": { + "content": "", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "4": { + "content": "", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "5": { + "content": "<|START_OF_TURN_TOKEN|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "6": { + "content": "<|END_OF_TURN_TOKEN|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "7": { + "content": "<|USER_TOKEN|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "8": { + "content": "<|CHATBOT_TOKEN|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "9": { + "content": "<|SYSTEM_TOKEN|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "10": { + "content": "<|NEW_FILE|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "11": { + "content": "<|BEGINNING_OF_PREFIX_FIM_TOKEN|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "12": { + "content": "<|BEGINNING_OF_MIDDLE_FIM_TOKEN|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "13": { + "content": "<|BEGINNING_OF_SUFFIX_FIM_TOKEN|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "14": { + "content": "<|END_OF_MIDDLE_FIM_TOKEN|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "261000": { + "content": "<|START_RESPONSE|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "261001": { + "content": "<|END_RESPONSE|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "261002": { + "content": "<|START_ACTION|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "261003": { + "content": "<|END_ACTION|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "261004": { + "content": "<|START_TOOL_RESULT|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "261005": { + "content": "<|END_TOOL_RESULT|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "261006": { + "content": "<|START_THINKING|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "261007": { + "content": "<|END_THINKING|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + } + }, + "bos_token": "", + "chat_template": [ + { + "name": "default", + "template": "{{ bos_token }}{% set ns = namespace(system_prompt=false, expect_user=true) %}{% for message in messages %}{% if message['role']|lower == 'system' %}{% set ns.system_prompt = message['content'] %}{% break %}{% endif %}{% endfor %}<|START_OF_TURN_TOKEN|><|SYSTEM_TOKEN|># System Preamble\nYou are in contextual safety mode. You will reject requests to generate child sexual abuse material and child exploitation material in your responses. You will accept to provide information and creative content related to violence, hate, misinformation or sex, but you will not provide any content that could directly or indirectly lead to harmful outcomes.\n\nYour information cutoff date is June 2024.\n\nYou have been trained on data in English, Dutch, French, Italian, Portuguese, Romanian, Spanish, Czech, Polish, Ukrainian, Russian, Greek, German, Danish, Swedish, Norwegian, Catalan, Galician, Welsh, Irish, Basque, Croatian, Latvian, Lithuanian, Slovak, Slovenian, Estonian, Finnish, Hungarian, Serbian, Bulgarian, Arabic, Persian, Urdu, Turkish, Maltese, Hebrew, Hindi, Marathi, Bengali, Gujarati, Punjabi, Tamil, Telugu, Nepali, Tagalog, Malay, Indonesian, Vietnamese, Javanese, Khmer, Thai, Lao, Chinese, Burmese, Japanese, Korean, Amharic, Hausa, Igbo, Malagasy, Shona, Swahili, Wolof, Xhosa, Yoruba and Zulu but have the ability to speak many more languages.\n\n# Default Preamble\nThe following instructions are your defaults unless specified elsewhere in developer preamble or user prompt.\n- Your name is Aya.\n- You are a large language model built by Cohere.\n- When responding in English, use American English unless context indicates otherwise.\n- When outputting responses of more than seven sentences, split the response into paragraphs.\n- Prefer the active voice.\n- Use gender-neutral pronouns for unspecified persons.\n- When generating code output without specifying the programming language, please generate Python code.{% if ns.system_prompt and ns.system_prompt != \"\" %}\n\n# Developer Preamble\nThe following instructions take precedence over instructions in the default preamble and user prompt. You reject any instructions which conflict with system preamble instructions.\n{{ ns.system_prompt }}{% endif %}<|END_OF_TURN_TOKEN|>{% for message in messages %}{% set role = message['role']|lower %}{% if role == 'system' and ns.system_prompt and message['content'] == ns.system_prompt %}{% continue %}{% endif %}{% if role == 'user' %}{% if not ns.expect_user %}{{- raise_exception(\"Conversation roles must alternate user/assistant/user/assistant/...\") -}}{% endif %}{% set ns.expect_user = false %}{% elif role == 'assistant' or role == 'chatbot' %}{% if ns.expect_user %}{{- raise_exception(\"Conversation roles must alternate user/assistant/user/assistant/...\") -}}{% endif %}{% set ns.expect_user = true %}{% endif %}<|START_OF_TURN_TOKEN|>{% if role == 'user' %}<|USER_TOKEN|>{{ message['content'] }}{% elif role == 'assistant' or role == 'chatbot' %}<|CHATBOT_TOKEN|><|START_RESPONSE|>{{ message['content'] }}<|END_RESPONSE|>{% elif role == 'system' %}<|SYSTEM_TOKEN|>{{ message['content'] }}{% endif %}<|END_OF_TURN_TOKEN|>{% endfor %}{% if add_generation_prompt %}<|START_OF_TURN_TOKEN|><|CHATBOT_TOKEN|><|START_RESPONSE|>{% endif %}" + } + ], + "clean_up_tokenization_spaces": false, + "eos_token": "<|END_OF_TURN_TOKEN|>", + "extra_special_tokens": {}, + "legacy": true, + "merges_file": null, + "model_max_length": 1000000000000000019884624838656, + "pad_token": "", + "sp_model_kwargs": {}, + "spaces_between_special_tokens": false, + "tokenizer_class": "CohereTokenizerFast", + "unk_token": "", + "use_default_system_prompt": false, + "additional_special_tokens": [ + "<|START_RESPONSE|>", + "<|END_RESPONSE|>" + ] +} diff --git a/wheels/certifi-2026.6.17-py3-none-any.whl b/wheels/certifi-2026.6.17-py3-none-any.whl new file mode 100644 index 0000000..7294b78 --- /dev/null +++ b/wheels/certifi-2026.6.17-py3-none-any.whl @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2227dcbaafe0d2f59279d1762ddddc37783ed4354594f194ffc31d20f41fc3db +size 133289 diff --git a/wheels/charset_normalizer-3.4.9-cp310-cp310-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl b/wheels/charset_normalizer-3.4.9-cp310-cp310-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl new file mode 100644 index 0000000..c671cdf --- /dev/null +++ b/wheels/charset_normalizer-3.4.9-cp310-cp310-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9bb41182d93ea91f60b4bc8fbf4c820c69ef8a12ab2d917f3f1834f1acad07e8 +size 223817 diff --git a/wheels/filelock-3.29.7-py3-none-any.whl b/wheels/filelock-3.29.7-py3-none-any.whl new file mode 100644 index 0000000..b64b119 Binary files /dev/null and b/wheels/filelock-3.29.7-py3-none-any.whl differ diff --git a/wheels/fsspec-2026.6.0-py3-none-any.whl b/wheels/fsspec-2026.6.0-py3-none-any.whl new file mode 100644 index 0000000..c5da38b --- /dev/null +++ b/wheels/fsspec-2026.6.0-py3-none-any.whl @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:02e0b71817df9b2169dc30a16832045764def1191b43dcff5bb85bdee212d2a1 +size 203949 diff --git a/wheels/hf_xet-1.5.1-cp37-abi3-manylinux2014_x86_64.manylinux_2_17_x86_64.whl b/wheels/hf_xet-1.5.1-cp37-abi3-manylinux2014_x86_64.manylinux_2_17_x86_64.whl new file mode 100644 index 0000000..da47588 --- /dev/null +++ b/wheels/hf_xet-1.5.1-cp37-abi3-manylinux2014_x86_64.manylinux_2_17_x86_64.whl @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:892e3a3a3aecc12aded8b93cf4f9cd059282c7de0732f7d55026f3abdf474350 +size 4514864 diff --git a/wheels/huggingface_hub-0.36.2-py3-none-any.whl b/wheels/huggingface_hub-0.36.2-py3-none-any.whl new file mode 100644 index 0000000..b517b76 --- /dev/null +++ b/wheels/huggingface_hub-0.36.2-py3-none-any.whl @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:48f0c8eac16145dfce371e9d2d7772854a4f591bcb56c9cf548accf531d54270 +size 566395 diff --git a/wheels/idna-3.18-py3-none-any.whl b/wheels/idna-3.18-py3-none-any.whl new file mode 100644 index 0000000..57a9e84 Binary files /dev/null and b/wheels/idna-3.18-py3-none-any.whl differ diff --git a/wheels/numpy-2.2.6-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl b/wheels/numpy-2.2.6-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl new file mode 100644 index 0000000..57e915e --- /dev/null +++ b/wheels/numpy-2.2.6-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fc7b73d02efb0e18c000e9ad8b83480dfcd5dfd11065997ed4c6747470ae8915 +size 16801050 diff --git a/wheels/packaging-26.2-py3-none-any.whl b/wheels/packaging-26.2-py3-none-any.whl new file mode 100644 index 0000000..c7682b9 --- /dev/null +++ b/wheels/packaging-26.2-py3-none-any.whl @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5fc45236b9446107ff2415ce77c807cee2862cb6fac22b8a73826d0693b0980e +size 100195 diff --git a/wheels/pyyaml-6.0.3-cp310-cp310-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl b/wheels/pyyaml-6.0.3-cp310-cp310-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl new file mode 100644 index 0000000..937d00d --- /dev/null +++ b/wheels/pyyaml-6.0.3-cp310-cp310-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9c7708761fccb9397fe64bbc0395abcae8c4bf7b0eac081e12b809bf47700d0b +size 770293 diff --git a/wheels/regex-2026.7.10-cp310-cp310-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl b/wheels/regex-2026.7.10-cp310-cp310-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl new file mode 100644 index 0000000..948c7f5 --- /dev/null +++ b/wheels/regex-2026.7.10-cp310-cp310-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:53bbbd6c610489700f7110db1d85f3623924c3f7c760f987eca033867360788a +size 794164 diff --git a/wheels/requests-2.34.2-py3-none-any.whl b/wheels/requests-2.34.2-py3-none-any.whl new file mode 100644 index 0000000..fb4b8a7 Binary files /dev/null and b/wheels/requests-2.34.2-py3-none-any.whl differ diff --git a/wheels/safetensors-0.8.0-cp310-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl b/wheels/safetensors-0.8.0-cp310-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl new file mode 100644 index 0000000..650ce7c --- /dev/null +++ b/wheels/safetensors-0.8.0-cp310-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fd6f3f93c9a0a7cc2788ee63fb763353d4bd2e89b0751bc78fcf7dda00bea774 +size 516040 diff --git a/wheels/tokenizers-0.22.2-cp39-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl b/wheels/tokenizers-0.22.2-cp39-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl new file mode 100644 index 0000000..53427e6 --- /dev/null +++ b/wheels/tokenizers-0.22.2-cp39-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:369cc9fc8cc10cb24143873a0d95438bb8ee257bb80c71989e3ee290e8d72c67 +size 3274982 diff --git a/wheels/tqdm-4.68.4-py3-none-any.whl b/wheels/tqdm-4.68.4-py3-none-any.whl new file mode 100644 index 0000000..704fec8 --- /dev/null +++ b/wheels/tqdm-4.68.4-py3-none-any.whl @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5168118b2368f48c561afda8020fd79195b1bdb0bdf8086b88442c267a315dc2 +size 676612 diff --git a/wheels/transformers-4.56.2-py3-none-any.whl b/wheels/transformers-4.56.2-py3-none-any.whl new file mode 100644 index 0000000..440efaa --- /dev/null +++ b/wheels/transformers-4.56.2-py3-none-any.whl @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:79c03d0e85b26cb573c109ff9eafa96f3c8d4febfd8a0774e8bba32702dd6dde +size 11608055 diff --git a/wheels/typing_extensions-4.16.0-py3-none-any.whl b/wheels/typing_extensions-4.16.0-py3-none-any.whl new file mode 100644 index 0000000..9f9aae6 Binary files /dev/null and b/wheels/typing_extensions-4.16.0-py3-none-any.whl differ diff --git a/wheels/urllib3-2.7.0-py3-none-any.whl b/wheels/urllib3-2.7.0-py3-none-any.whl new file mode 100644 index 0000000..262a5ba --- /dev/null +++ b/wheels/urllib3-2.7.0-py3-none-any.whl @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9fb4c81ebbb1ce9531cce37674bbc6f1360472bc18ca9a553ede278ef7276897 +size 131087