初始化项目,由ModelHub XC社区提供模型

Model: Polygl0t/Tucano2-qwen-1.5B-Think
Source: Original Platform
This commit is contained in:
ModelHub XC
2026-06-19 18:17:12 +08:00
commit 054f571835
22 changed files with 465596 additions and 0 deletions

41
.gitattributes vendored Normal file
View File

@@ -0,0 +1,41 @@
*.7z filter=lfs diff=lfs merge=lfs -text
*.arrow filter=lfs diff=lfs merge=lfs -text
*.bin filter=lfs diff=lfs merge=lfs -text
*.bz2 filter=lfs diff=lfs merge=lfs -text
*.ckpt filter=lfs diff=lfs merge=lfs -text
*.ftz filter=lfs diff=lfs merge=lfs -text
*.gz filter=lfs diff=lfs merge=lfs -text
*.h5 filter=lfs diff=lfs merge=lfs -text
*.joblib filter=lfs diff=lfs merge=lfs -text
*.lfs.* filter=lfs diff=lfs merge=lfs -text
*.mlmodel filter=lfs diff=lfs merge=lfs -text
*.model filter=lfs diff=lfs merge=lfs -text
*.msgpack filter=lfs diff=lfs merge=lfs -text
*.npy filter=lfs diff=lfs merge=lfs -text
*.npz filter=lfs diff=lfs merge=lfs -text
*.onnx filter=lfs diff=lfs merge=lfs -text
*.ot filter=lfs diff=lfs merge=lfs -text
*.parquet filter=lfs diff=lfs merge=lfs -text
*.pb filter=lfs diff=lfs merge=lfs -text
*.pickle filter=lfs diff=lfs merge=lfs -text
*.pkl filter=lfs diff=lfs merge=lfs -text
*.pt filter=lfs diff=lfs merge=lfs -text
*.pth filter=lfs diff=lfs merge=lfs -text
*.rar filter=lfs diff=lfs merge=lfs -text
*.safetensors filter=lfs diff=lfs merge=lfs -text
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
*.tar.* filter=lfs diff=lfs merge=lfs -text
*.tar filter=lfs diff=lfs merge=lfs -text
*.tflite filter=lfs diff=lfs merge=lfs -text
*.tgz filter=lfs diff=lfs merge=lfs -text
*.wasm filter=lfs diff=lfs merge=lfs -text
*.xz filter=lfs diff=lfs merge=lfs -text
*.zip filter=lfs diff=lfs merge=lfs -text
*.zst filter=lfs diff=lfs merge=lfs -text
*tfevents* filter=lfs diff=lfs merge=lfs -text
.plots/apo_gradient_norm.png filter=lfs diff=lfs merge=lfs -text
.plots/apo_reward.png filter=lfs diff=lfs merge=lfs -text
.plots/model_comparison.png filter=lfs diff=lfs merge=lfs -text
.plots/sft_gradient_norm.png filter=lfs diff=lfs merge=lfs -text
.plots/sft_loss.png filter=lfs diff=lfs merge=lfs -text
logo.png filter=lfs diff=lfs merge=lfs -text

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:29bd25077d0963148cb0f4037f0cb8dc118a541def9d889dfefddb3f8aeb09de
size 482994

3
.plots/apo_reward.png Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:3ab2d18bdb9961ad8a8bc0e7274735ebb2c337e2ed2cc389b3b00f81c1e66f7e
size 286839

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:d798bab25626567c5d6b038d5631782671b04b10684a06daaff40297925e01a8
size 217771

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:0de46cb25883cd57d12cb8e6fa41c6e810a209b7ce2e3d797ca5f41831a9f3da
size 318225

3
.plots/sft_loss.png Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:c38664279e9165a2b64c2e062894747b34f55953834b96c7613089c0000d39fe
size 383372

190
LICENSE Normal file
View File

@@ -0,0 +1,190 @@
Apache License
Version 2.0, January 2004
http://www.apache.org/licenses/
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
1. Definitions.
"License" shall mean the terms and conditions for use, reproduction,
and distribution as defined by Sections 1 through 9 of this document.
"Licensor" shall mean the copyright owner or entity authorized by
the copyright owner that is granting the License.
"Legal Entity" shall mean the union of the acting entity and all
other entities that control, are controlled by, or are under common
control with that entity. For the purposes of this definition,
"control" means (i) the power, direct or indirect, to cause the
direction or management of such entity, whether by contract or
otherwise, or (ii) ownership of fifty percent (50%) or more of the
outstanding shares, or (iii) beneficial ownership of such entity.
"You" (or "Your") shall mean an individual or Legal Entity
exercising permissions granted by this License.
"Source" form shall mean the preferred form for making modifications,
including but not limited to software source code, documentation
source, and configuration files.
"Object" form shall mean any form resulting from mechanical
transformation or translation of a Source form, including but
not limited to compiled object code, generated documentation,
and conversions to other media types.
"Work" shall mean the work of authorship, whether in Source or
Object form, made available under the License, as indicated by a
copyright notice that is included in or attached to the work
(an example is provided in the Appendix below).
"Derivative Works" shall mean any work, whether in Source or Object
form, that is based on (or derived from) the Work and for which the
editorial revisions, annotations, elaborations, or other modifications
represent, as a whole, an original work of authorship. For the purposes
of this License, Derivative Works shall not include works that remain
separable from, or merely link (or bind by name) to the interfaces of,
the Work and Derivative Works thereof.
"Contribution" shall mean any work of authorship, including
the original version of the Work and any modifications or additions
to that Work or Derivative Works thereof, that is intentionally
submitted to Licensor for inclusion in the Work by the copyright owner
or by an individual or Legal Entity authorized to submit on behalf of
the copyright owner. For the purposes of this definition, "submitted"
means any form of electronic, verbal, or written communication sent
to the Licensor or its representatives, including but not limited to
communication on electronic mailing lists, source code control systems,
and issue tracking systems that are managed by, or on behalf of, the
Licensor for the purpose of discussing and improving the Work, but
excluding communication that is conspicuously marked or otherwise
designated in writing by the copyright owner as "Not a Contribution."
"Contributor" shall mean Licensor and any individual or Legal Entity
on behalf of whom a Contribution has been received by Licensor and
subsequently incorporated within the Work.
2. Grant of Copyright License. Subject to the terms and conditions of
this License, each Contributor hereby grants to You a perpetual,
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
copyright license to reproduce, prepare Derivative Works of,
publicly display, publicly perform, sublicense, and distribute the
Work and such Derivative Works in Source or Object form.
3. Grant of Patent License. Subject to the terms and conditions of
this License, each Contributor hereby grants to You a perpetual,
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
(except as stated in this section) patent license to make, have made,
use, offer to sell, sell, import, and otherwise transfer the Work,
where such license applies only to those patent claims licensable
by such Contributor that are necessarily infringed by their
Contribution(s) alone or by combination of their Contribution(s)
with the Work to which such Contribution(s) was submitted. If You
institute patent litigation against any entity (including a
cross-claim or counterclaim in a lawsuit) alleging that the Work
or a Contribution incorporated within the Work constitutes direct
or contributory patent infringement, then any patent licenses
granted to You under this License for that Work shall terminate
as of the date such litigation is filed.
4. Redistribution. You may reproduce and distribute copies of the
Work or Derivative Works thereof in any medium, with or without
modifications, and in Source or Object form, provided that You
meet the following conditions:
(a) You must give any other recipients of the Work or
Derivative Works a copy of this License; and
(b) You must cause any modified files to carry prominent notices
stating that You changed the files; and
(c) You must retain, in the Source form of any Derivative Works
that You distribute, all copyright, patent, trademark, and
attribution notices from the Source form of the Work,
excluding those notices that do not pertain to any part of
the Derivative Works; and
(d) If the Work includes a "NOTICE" text file as part of its
distribution, then any Derivative Works that You distribute must
include a readable copy of the attribution notices contained
within such NOTICE file, excluding those notices that do not
pertain to any part of the Derivative Works, in at least one
of the following places: within a NOTICE text file distributed
as part of the Derivative Works; within the Source form or
documentation, if provided along with the Derivative Works; or,
within a display generated by the Derivative Works, if and
wherever such third-party notices normally appear. The contents
of the NOTICE file are for informational purposes only and
do not modify the License. You may add Your own attribution
notices within Derivative Works that You distribute, alongside
or as an addendum to the NOTICE text from the Work, provided
that such additional attribution notices cannot be construed
as modifying the License.
You may add Your own copyright statement to Your modifications and
may provide additional or different license terms and conditions
for use, reproduction, or distribution of Your modifications, or
for any such Derivative Works as a whole, provided Your use,
reproduction, and distribution of the Work otherwise complies with
the conditions stated in this License.
5. Submission of Contributions. Unless You explicitly state otherwise,
any Contribution intentionally submitted for inclusion in the Work
by You to the Licensor shall be under the terms and conditions of
this License, without any additional terms or conditions.
Notwithstanding the above, nothing herein shall supersede or modify
the terms of any separate license agreement you may have executed
with Licensor regarding such Contributions.
6. Trademarks. This License does not grant permission to use the trade
names, trademarks, service marks, or product names of the Licensor,
except as required for reasonable and customary use in describing the
origin of the Work and reproducing the content of the NOTICE file.
7. Disclaimer of Warranty. Unless required by applicable law or
agreed to in writing, Licensor provides the Work (and each
Contributor provides its Contributions) on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
implied, including, without limitation, any warranties or conditions
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
PARTICULAR PURPOSE. You are solely responsible for determining the
appropriateness of using or redistributing the Work and assume any
risks associated with Your exercise of permissions under this License.
8. Limitation of Liability. In no event and under no legal theory,
whether in tort (including negligence), contract, or otherwise,
unless required by applicable law (such as deliberate and grossly
negligent acts) or agreed to in writing, shall any Contributor be
liable to You for damages, including any direct, indirect, special,
incidental, or consequential damages of any character arising as a
result of this License or out of the use or inability to use the
Work (including but not limited to damages for loss of goodwill,
work stoppage, computer failure or malfunction, or any and all
other commercial damages or losses), even if such Contributor
has been advised of the possibility of such damages.
9. Accepting Warranty or Additional Liability. While redistributing
the Work or Derivative Works thereof, You may choose to offer,
and charge a fee for, acceptance of support, warranty, indemnity,
or other liability obligations and/or rights consistent with this
License. However, in accepting such obligations, You may act only
on Your own behalf and on Your sole responsibility, not on behalf
of any other Contributor, and only if You agree to indemnify,
defend, and hold each Contributor harmless for any liability
incurred by, or claims asserted against, such Contributor by reason
of your accepting any such warranty or additional liability.
END OF TERMS AND CONDITIONS
Copyright Nicholas Kluge Corrêa, Shiza Fatimah, Aniket Sen, and Sophia Falk
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.

531
README.md Normal file
View File

@@ -0,0 +1,531 @@
---
language:
- pt
license: apache-2.0
library_name: transformers
tags:
- text-generation-inference
datasets:
- Polygl0t/gigaverbo-v2-sft
- Polygl0t/gigaverbo-v2-preferences
metrics:
- perplexity
pipeline_tag: text-generation
widget:
- text: "<|im_start|>user\nQual é a capital de Portugal?<|im_end|><|im_start|>assistant\n"
example_title: Exemplo
- text: "<|im_start|>user\nEscreva um poema sobre a floresta amazônica.<|im_end|><|im_start|>assistant\n"
example_title: Exemplo
- text: "<|im_start|>user\nListe três benefícios da energia solar.<|im_end|><|im_start|>assistant\n"
example_title: Exemplo
inference:
parameters:
repetition_penalty: 1.2
temperature: 0.1
top_k: 50
top_p: 1.0
max_new_tokens: 150
co2_eq_emissions:
emissions: 2550
source: CodeCarbon
training_type: post-training
geographical_location: Germany
hardware_used: NVIDIA A100-SXM4-80GB
model-index:
- name: Tucano2-qwen-1.5B-Think
results:
- task:
type: text-generation
name: Text Generation
dataset:
name: ARC Challenge
type: Polygl0t/ARC-poly
split: test
args:
num_few_shot: 5
metrics:
- type: acc_norm
value: 42.82
name: Acc-norm
source:
url: https://github.com/Polygl0t/lm-evaluation-harness/tree/polyglot_harness_portuguese
name: arc_challenge_poly_pt
- task:
type: text-generation
name: Text Generation
dataset:
name: MMLU (Portuguese)
type: Polygl0t/MMLU-poly
split: test
args:
num_few_shot: 5
metrics:
- type: acc
value: 43.3
name: Acc
source:
url: https://github.com/Polygl0t/lm-evaluation-harness/tree/polyglot_harness_portuguese
name: mmlu_poly_pt
- task:
type: text-generation
name: Text Generation
dataset:
name: BELEBELE (Portuguese)
type: facebook/belebele
split: test
args:
num_few_shot: 5
metrics:
- type: acc_norm
value: 67.67
name: Acc-norm
source:
url: https://github.com/Polygl0t/lm-evaluation-harness/tree/polyglot_harness_portuguese
name: belebele_pot_Latn
- task:
type: text-generation
name: Text Generation
dataset:
name: BLUEX
type: eduagarcia-temp/BLUEX_without_images
split: train
args:
num_few_shot: 3
metrics:
- type: acc
value: 39.22
name: Acc
source:
url: https://github.com/eduagarcia/lm-evaluation-harness-pt
name: bluex
- task:
type: text-generation
name: Text Generation
dataset:
name: ENEM Challenge
type: eduagarcia/enem_challenge
split: train
args:
num_few_shot: 3
metrics:
- type: acc
value: 39.89
name: Acc
source:
url: https://github.com/eduagarcia/lm-evaluation-harness-pt
name: enem_challenge
- task:
type: text-generation
name: Text Generation
dataset:
name: OAB Exams
type: eduagarcia/oab_exams
split: train
args:
num_few_shot: 3
metrics:
- type: acc
value: 34.26
name: Acc
source:
url: https://github.com/eduagarcia/lm-evaluation-harness-pt
name: oab_exams
- task:
type: text-generation
name: Text Generation
dataset:
name: IFEval
type: Polygl0t/IFEval-PT
split: train
args:
num_few_shot: 0
metrics:
- type: ifeval_pt_prompt_level_loose_acc
value: 33.67
name: Acc-loose
source:
url: https://github.com/Polygl0t/lm-evaluation-harness/tree/polyglot_harness_portuguese
name: ifeval_pt
- task:
type: text-generation
name: Text Generation
dataset:
name: GSM8K
type: Polygl0t/gsm8k-pt
split: test
args:
num_few_shot: 0
metrics:
- type: flexible-extract
value: 22.83
name: Acc-flex
source:
url: https://github.com/Polygl0t/lm-evaluation-harness/tree/polyglot_harness_portuguese
name: gsm8k_pt
base_model: Polygl0t/Tucano2-qwen-1.5B-Base
---
# Tucano2-qwen-1.5B-Think
<img src="./logo.png" alt="An illustration of a Tucano bird showing vibrant colors like yellow, orange, blue, green, and black." height="200">
## Model Summary
**[Tucano2-qwen-1.5B-Think](https://huggingface.co/Polygl0t/Tucano2-qwen-1.5B-Think)** is an instruction-tuned Portuguese language model built on top of **Tucano2-qwen-0.5B-Base**. It has been trained using a combination of one round of supervised fine-tuning (SFT) and one round of Anchored Preference Optimization (APO).
Tucano2-qwen-1.5B-Think is a reasoning model, which means it has been fine-tuned to generate CoT-style (Chain-of-Thought) traces in its responses. These reasoning traces are always encapsulated within the special tokens `<think>` and `</think>`.
**All datasets, source code, and training recipes used to develop the Tucano2 series are fully open and reproducible.**
## Details
- **Architecture:** a Transformer-based model ([`qwen3`](https://huggingface.co/docs/transformers/main/en/model_doc/qwen3))
- **Size:** 1,510,073,344 parameters
- **Context length:** 4,096 tokens
- **Dataset(s):**
- [Polygl0t/gigaverbo-v2-sft](https://huggingface.co/datasets/Polygl0t/gigaverbo-v2-sft)
- [Polygl0t/gigaverbo-v2-preferences](https://huggingface.co/datasets/Polygl0t/gigaverbo-v2-preferences)
- **Training time**: ~ 1.5 hours
- **Emissions:** 2.55 KgCO2 (Germany)
- **Total energy consumption:** 5.5 kWh
This repository has the [source code](https://github.com/Polygl0t/llm-foundry) used to train this model. The full configuration used for training is available in the following config files:
- Single stage Supervised Fine-Tuning (linear warmup with cosine decay): [training_config_sft.yaml](training_config_sft.yaml)
- Single stage Anchored Preference Optimization (linear warmup with cosine decay): [training_config_apo.yaml](training_config_apo.yaml)
- Training Logs (loss, lr, rewards, etc.): [train_logs_apo.parquet](train_logs_apo.parquet), [train_logs_sft.parquet](train_logs_sft.parquet)
<details>
<summary><b>SFT Loss Curve</b></summary>
![SFT Loss Curve](./.plots/sft_loss.png)
</details>
<details>
<summary><b>APO Rewards</b></summary>
![APO Rewards](./.plots/apo_reward.png)
</details>
## Intended Uses
The primary intended use Tucano2-qwen-1.5B-Think is to serve as foundations for research and development involving Portuguese language modeling. You may also fine-tune and adapt Tucano2-qwen-1.5B-Think for deployment if your use follows the Apache 2.0 license. If you decide to use Tucano2-qwen-1.5B-Think as a basis for your fine-tuned model, please conduct your own risk and bias assessment.
## Basic usage
```python
from transformers import AutoTokenizer, AutoModelForCausalLM, GenerationConfig
import torch
# Load model and tokenizer
model_id = "Polygl0t/Tucano2-qwen-1.5B-Think"
tokenizer = AutoTokenizer.from_pretrained(model_id)
model = AutoModelForCausalLM.from_pretrained(
model_id,
device_map="auto"
)
# Configure generation parameters
generation_config = GenerationConfig(
do_sample=True,
temperature=0.1,
top_k=50,
top_p=1.0,
repetition_penalty=1.2,
max_new_tokens=150,
pad_token_id=tokenizer.eos_token_id,
)
# Prepare chat messages
messages = [
{"role": "user", "content": "Qual é a capital de Portugal"}
]
# Apply chat template and generate
prompt = tokenizer.apply_chat_template(messages, tokenize=False, add_generation_prompt=True)
inputs = tokenizer(prompt, return_tensors="pt").to(model.device)
with torch.no_grad():
outputs = model.generate(**inputs, generation_config=generation_config)
# Decode and print response
full_output = tokenizer.decode(
outputs[0][len(inputs.input_ids[0]):],
skip_special_tokens=True
).strip()
# Extract <think>...</think> content
think_content = None
final_response = full_output
if "<think>" in full_output and "</think>" in full_output:
start = full_output.find("<think>") + len("<think>")
end = full_output.find("</think>")
think_content = full_output[start:end].strip()
# Remove think block from final response
final_response = (
full_output[:full_output.find("<think>")] +
full_output[end + len("</think>"):]
).strip()
if think_content:
print("🧠 Thinking:\n")
print(think_content)
print("\n" + "="*50 + "\n")
print("🤖 Answer:\n")
print(final_response)
```
## Limitations
Like almost all other language models trained on large text datasets scraped from the web, the Tucano2-qwen-1.5B-Think shows behavior that does not make it an out-of-the-box solution to many real-world applications, especially those requiring factual, reliable, and nontoxic text generation. Tucano2-qwen-1.5B-Think is subject to the following:
- **Hallucinations:** Tucano2-qwen-1.5B-Think can produce content that can be mistaken as true facts, but are misleading or entirely false, i.e., hallucination.
- **Biases and Toxicity:** Tucano2-qwen-1.5B-Think inherits the social and historical stereotypes from the data used to train it. Given these biases, the model can produce toxic content, i.e., harmful, offensive, or detrimental to individuals, groups, or communities.
- **Language Limitations:** Tucano2-qwen-1.5B-Think is primarily designed to interact with Portuguese. Other languages might challenge its comprehension, leading to potential misinterpretations or errors in response.
- **Repetition and Verbosity:** Tucano2-qwen-1.5B-Think may get stuck on repetition loops (especially if the repetition penalty during generations is set to a meager value) or produce verbose responses unrelated to the prompt it was given.
Hence, even though Tucano2-qwen-1.5B-Think is released with a permissive license, we urge users to perform their risk analysis on them if they intend to use them for real-world applications.
## Evaluations
The table below compares the Tucano2 (Think variant) series against other reasoning models of similar size. We divide our evaluations into two sets:
- **Knowledge & Reasoning:** ARC-Challenge, ENEM, BLUEX, OAB Exams, BELEBELE, MMLU, GSM8K-PT
- **Instruction Following:** IFEval-PT
The NPM (Normalized Performance Metric) provides a balanced view of model performance across tasks, accounting for each task's inherent difficulty by normalizing its evaluation score relative to its random baseline.
We do not include coding benchmarks in this table because the Think models were not trained on coding data during post-training and thus perform poorly on them. For coding skills, we recommend using the Instruct models instead, which were trained with coding data and perform much better on coding benchmarks.
| | Total Avg. | Knowledge & Reasoning (NPM) | Instruction Following |
| --------------------------- | ---------- | --------------------------- | --------------------- |
| **Tucano2-qwen-3.7B-Think** | 51.27 | 54.07 | 31.67 |
| SmolLM3-3B | 48.58 | 46.28 | 64.67 |
| Qwen3-4B | 46.35 | 40.97 | 84 |
| Qwen3-1.7B | 36.54 | 32 | 68.33 |
| **Tucano2-qwen-1.5B-Think** | 27.54 | 26.67 | 33.67 |
| Qwen3-0.6B | 24.11 | 19.22 | 58.33 |
| **Tucano2-qwen-0.5B-Think** | 14.41 | 12.52 | 27.67 |
<details>
<summary><b>Evaluation Suite</b></summary>
| **Benchmark** | **n-shot** | **Type** | **Baseline** | **Metric** |
| ------------------------- | ---------- | ------------- | ------------ | ------------------------ |
| **Knowledge & Reasoning** | | | | |
| ARC-Challenge | 5-shot | MC-Q&A | 25 | `acc_norm` |
| ENEM | 3-shot | MC-Q&A | 20 | `acc` |
| BLUEX | 3-shot | MC-Q&A | 22.5 | `acc` |
| OAB Exams | 3-shot | MC-Q&A | 25 | `acc` |
| BELEBELE | 5-shot | MC-Q&A | 25 | `acc_norm` |
| MMLU | 5-shot | MC-Q&A | 25 | `acc` |
| GSM8K-PT | 0-shot | Math Problems | 0 | `flexible-extract` |
| **Instruction Following** | | | | |
| IFEval-PT | 0-shot | Instruction | 0 | `prompt_level_loose_acc` |
</details>
<details>
<summary><b>Individual Benchmarks</b></summary>
| | BLUEX | ENEM | OAB | ARC Challenge | BELEBELE | MMLU | IFEval-PT | GSM8K-PT |
| --------------------------- | ----- | ----- | ----- | ------------- | -------- | ----- | --------- | -------- |
| **Tucano2-qwen-3.7B-Think** | 63 | 71.52 | 53.76 | 55.38 | 83.56 | 61.18 | 31.67 | 56.7 |
| SmolLM3-3B | 48.82 | 60.6 | 42.19 | 52.56 | 77.78 | 52.82 | 64.67 | 72.15 |
| Qwen3-4B | 78.3 | 85.23 | 47.97 | 39.66 | 23 | 31.53 | 84 | 77.02 |
| Qwen3-1.7B | 68.29 | 72.5 | 39.32 | 35.64 | 22.89 | 22.85 | 68.33 | 71.69 |
| **Tucano2-qwen-1.5B-Think** | 39.22 | 39.89 | 34.26 | 42.82 | 67.67 | 43.3 | 33.67 | 22.83 |
| Qwen3-0.6B | 43.53 | 48.71 | 34.21 | 29.91 | 22.89 | 22.85 | 58.33 | 58.37 |
| **Tucano2-qwen-0.5B-Think** | 34.49 | 31.98 | 27.02 | 32.74 | 36.11 | 36.08 | 27.67 | 14.61 |
</details>
<details>
<summary><b>Inference Samples on Various Tasks</b></summary>
### Cooking
**Prompt:**
<|im_start|>user
Você pode me dar uma receita simples de Bolo de Brigadeiro?<|im_end|>
<|im_start|>assistant
**Response:**
<think>
O usuário está me pedindo uma receita simplificada de bolo de brigadeiro. Primeiro, vou analisar a solicitação para entender exatamente o que ele precisa. Preciso garantir que a receita seja fácil e rápida de seguir, então decido omitir alguns passos técnicos mais complexos. Agora preciso considerar os ingredientes básicos necessários para fazer um bom bolo de brigadeiro. Vou listar todos eles na minha mente: farinha de trigo, açúcar refinado, ovos grandes, óleo vegetal, leite condensado, chocolate amargo picado e gelatina em pó sem sabor. Devo organizar esses itens numa lista clara e concisa. Além disso, é importante mencionar as etapas principais da preparação do recheio (gelatina dissolvida no leite) e dos dois bolos individuais (farinha peneirada misturada com água morna). Também devo sugerir como untar e enfarinhar formas retangulares pequenas antes de colocar os bolos dentro delas. Finalmente, adicionarei instruções rápidas sobre assar nos forno pré-aquecido e resfriá-los completamente após retirar das formas. Antes de concluir, ofereço ao usuário a opção de pedir dicas extras caso precise. Isso mostra disponibilidade e ajuda quem possa estar buscando informações adicionais.
</think>
Claro! Aqui vai uma versão bem simplificada de **Bolo de Brigadeiro** que você pode preparar rapidamente:
### Ingredientes:
- 2 xícaras de farinha de trigo
- 1 xícara de açúcar refinado
- 4 ovos grandes
- ½ colher de chá de fermento químico incolor
- ⅓ copo de óleo vegetal
- 3 copos de leite condensado
- ₁ lata de chocolate amargo picado (aproximadamente 240g)
- Gelatina em pó sem sabor (1 tablete)
### Modo de Preparo:
1. Prepare a forma: Untie e dusted duas formas retangulares pequenos ou use papel manteiga.
2. Pré-aqueça o forno a 180°C.
3. Em uma panela pequena, derreta metade da gelatina em pó num pouco de água quente. Reserve.
4. Misture a outra metade da gelatina não derretida junto com todo o resto da geléia nas formas untadas. Deixe descansar por cerca de 5 minutos até ficar consistente.
5. Na batedeira, batam os ovos com o açúcar até obter uma mistura cremosa. Adicione gradualmente a farinha peneirada e mexendo apenas suficiente para combinar.
6. Acrescente o fermento químico aos poucos à massa já formada.
7. Gradualmente, incorpore o óleo vegetal às massas secas.
8. Por último, junte o leite condensado e o chocolate amargo picado. Mexer suavemente só pra misturar tudo uniformemente.
9. Preencha cada cavidade das formas preparadas anteriormente com esta mistura.
10. Asse durante 25-30 minutos aproximadamente, verificando se estão dourados pelo lado externo usando palitos finos.
11. Retire do forno e deixe esfriando totalmente dentro das formas.
12. Transfira cuidadosamente para bandejas de gelo cortáveis enquanto ainda estiverem quentes mas manejáveis.
13. Sirva fresco logo depois de desconectar do congelador.
E pronto! Um delicioso Bolinho de Brigadeiros caseiro rapidinho. Espero que goste! Qualquer coisa, estou aqui pra ajudar!<|im_end|>
---
### Math
**Prompt:**
<|im_start|>user
Como eu posso resolver o seguinte problema: 2x + 3 = 11?<|im_end|>
<|im_start|>assistant
**Response:**
<think>
O usuário está me perguntando como resolve-se a equação 2x + 3 = 11. Primeiro, vou analisar cuidadosamente cada elemento da equação. Preciso isolar x no lado esquerdo para descobrir seu valor. Vou começar subtraindo 3 de ambos os lados para eliminar esse termo constante do lado esquerdo. Isso dá duas vezes X igual ao número menos três e onze juntos. Agora tenho que dividir tudo por dois porque quero deixar só X multiplicado pelo coeficiente 2. Deixe-me fazer essa divisão agora. Dividindo 20/2 obtenho 10, então minha solução é x=10. Mas antes de confirmar isso, preciso revisitar todos os passos pra garantir nada foi deixado escapar ou cometido erro algum na manipulação algébrica. Sim, parece correto! Então posso dizer com confiança que a resposta correta é realmente 10.
</think>
Claro! Vamos resolver a equação \( 2x + 3 = 11 \) passo a passo.
### Passo 1: Isolar o termo com a variável \( x \)
Primeiro, vamos remover o termo independente (que não contém \( x \)) do lado onde estamos guardando nossa incógnita. Para isso, subtraímos 3 de ambos os lados da equação:
\[
2x + 3 - 3 = 11 - 3 \\
2x = 8
\]
Agora temos apenas \( 2x = 8 \).
### Passo 2: Resolver para \( x \)
Para encontrar o valor de \( x \), precisamos isolá-lo dividindo ambos os lados pela constante que multiplica \( x \):
\[
\frac{2x}{2} = \frac{8}{2}\quad (\text{dividimos ambos os lados por } 2)\\
x=4
\]
Portanto, a solução para a equação \( 2x + 3 = 11 \) é \( x = 4 \).
Se você tiver alguma outra dúvida ou quiser ver outro exemplo, estou aqui para ajudar!<|im_end|>
---
### Rewriting
**Prompt:**
<|im_start|>user
Remova todas as informações irrelevantes da frase a seguir.
"O novo restaurante que abriu no centro da cidade, de propriedade do primo de Jorge, que costumava ser chef de um restaurante em Ipanema, serve uma variedade de cozinhas de todo o mundo."<|im_end|>
<|im_start|>assistant
**Response:**
<think>
O usuário está me pedindo para remover toda informação desnecessária da frase fornecida. Primeiro, vou ler cuidadosamente a frase completa: 'O novo restaurante que abriu no centro da cidade, de propriedade do primo de Jorge, que costumava ser chefe de um restaurante em Ipanema, serve uma variedade de cozinhas de todo o mundo.' Agora preciso identificar quais partes são realmente importantes e relevantes ao objetivo principal dessa pergunta — entender qual tipo específico de cozinha esse restaurante oferece. Vou começar eliminando detalhes secundários ou redundâncias.
</think>
Entendi! Vamos focar nas informações essenciais sobre o tipo de culinária oferecida pelo restaurante:
> O novo restaurante no centro da cidade serve uma variedade de cozinhas de todo o mundo.
Se você quiser alguma outra coisa específica, estou aqui pra ajudar!<|im_end|>
---
### History
**Prompt:**
<|im_start|>user
Liste os principais eventos envolvendo a Revolução Farroupilha.<|im_end|>
<|im_start|>assistant
**Response:**
<think>
O usuário está me perguntando para listar os principais eventos envolvidos na Revolução Farroupilha. Primeiro, vou analisar exatamente o que ele quer saber: são fatos cronológicos importantes e significativos dessa importante revolta gaúcha? Agora preciso lembrar dos detalhes históricos relevantes da época. Vou começar pelo início do movimento, pois isso dá contexto ao leitor sobre quando tudo começou. Devo mencionar as razões iniciais pelas quais surgiu esse conflito entre republicanos farrapos e autoridades imperiais. Depois disso, posso organizar minha lista em ordem temporal, destacando cada evento crucial. Preciso garantir que aborde tanto aspectos políticos quanto sociais-econômicos, já que esses foram elementos cruciais nos motivadores das revoltas locais. Também devo incluir datas aproximadas onde possível para facilitar a compreensão do tempo histórico. Finalmente, concluirei com uma nota geral sobre o impacto duradouro desta revolução no Rio Grande do Sul moderno. Isso ajudará o usuário a ter uma visão completa e contextualizada.
</think>
Claro! A Revolução Farroupilha, também conhecida como Guerra dos Farrapos ou Revolta dos Sargentos, foi um levante militar contra o Império Brasileiro ocorrido nas regiões sulistas atuais de Santa Catarina, Rio Grande do Sul e parte de São Paulo, durante o período imperial (1835-1845). Aqui estão alguns dos principais eventos:
1. **Início da Rebelião**:
- Em março de 1831, eclodiu a primeira insurreição liderada por Bento Gonçalves, ex-soldado desertor do exército regular brasileiro. Inicialmente, esta rebelião visava derrubar o governo centralizado sob dom Pedro I. No entanto, após algumas vitórias militares, ela se transformou numa luta pela autonomia provincial.
2. **Formação das Províncias Livres**:
- Com apoio popular crescente, grupos regionais formaram suas próprias forças armadas independentes chamadas "farrapos". Estas províncias livres incluíram Rio Grande (atual RS), Santa Catarina e Paraná. O líder máximo desses movimentos era Manuel Marques Lisboa, futuro marquês de Alegrete.
3. **Conflito Militar Principal**:
- As hostilidades começaram oficialmente em junho de 1836, marcando o começo formal da guerra civil rio-grandense. Os farrapos enfrentaram tropas imperiais lideradas inicialmente pelos generais David Canabarro e Félix Zavala. Apesar das dificuldades logísticas e numéricas, eles conseguiram manter controle significativo sobre áreas estratégicas.
4. **Negociação e Trégua Temporária**:
- Nos anos seguintes, houve períodos alternados de combates intensos seguidos por acordos negociados visando aliviar tensões temporariamente. Um exemplo notório é a assinatura do Tratado de Ponche Verde em outubro de 1841, embora tenha sido rompido rapidamente depois dele.
5. **Finalização do Conflito**:
- Após várias campanhas bem-sucedidas conduzidas principalmente pelo general José Antônio Correia de Melo, finalmente resultou na queda definitiva de Porto Alegre às mãos imperiais em maio de 1844, encerrando assim efetivamente a maior rebeldia regionalista até então registrada dentro fronteiras brasileiras formais.
6. **Impacto Político e Social**:
- Este conflito teve consequências profundas não só política mas socialmente. Resultou na criação legal da Província do Rio Grande do Sul através Lei Complementar nº 2, assinada logo após término bélico; além disto contribuiu decisivamente formação identidade nacional brasileira questionada muitos intelectuais contemporâneos revolucionários.
Esses pontos resumem brevemente alguns momentos chave daquela história rica e complexa. Espero que ajude você entender melhor este episódio fascinante da nossa herança cultural!<|im_end|>
---
</details>
Below, we compare the performance of Tucano2-qwen-1.5B-Think with Qwen3-1.7B, a strong baseline in the 1.5B parameter range. The percentages represent the absolute difference in performance between the two models on each benchmark. All other plots can be found in the [.plots](https://huggingface.co/Polygl0t/Tucano2-qwen-1.5B-Think/tree/main/.plots/) folder.
**Tucano2-qwen-1.5B-Think vs Qwen3-1.7B**
![Performance Comparison](./.plots/model_comparison.png)
## Cite as 🤗
```latex
@misc{correa2026tucano2cool,
title={{Tucano 2 Cool: Better Open Source LLMs for Portuguese}},
author={Nicholas Kluge Corr{\^e}a and Aniket Sen and Shiza Fatimah and Sophia Falk and Lennard Landgraf and Julia Kastner and Lucie Flek},
year={2026},
eprint={2603.03543},
archivePrefix={arXiv},
primaryClass={cs.CL},
url={https://arxiv.org/abs/2603.03543},
}
```
## Aknowlegments
Polyglot is a project funded by the Federal Ministry of Education and Research (BMBF) and the Ministry of Culture and Science of the State of North Rhine-Westphalia (MWK) as part of TRA Sustainable Futures (University of Bonn) and the Excellence Strategy of the federal and state governments.
We also gratefully acknowledge the granted access to the [Marvin cluster](https://www.hpc.uni-bonn.de/en/systems/marvin) hosted by [University of Bonn](https://www.uni-bonn.de/en) along with the support provided by its High Performance Computing & Analytics Lab.
## License
Tucano2-qwen-1.5B-Think is licensed under the Apache License, Version 2.0. For more details, see the [LICENSE](LICENSE) file.

114
chat_template.jinja Normal file
View File

@@ -0,0 +1,114 @@
{#- Handle tool/function calling setup #}
{%- if tools %}
{{- '<|im_start|>system\n' }}
{#- Include system message if present #}
{%- if messages[0].role == 'system' %}
{{- messages[0].content + '\n\n' }}
{%- endif %}
{#- Add tool calling instructions in Portuguese #}
{{- "# Tools / Ferramentas\n\nVocê pode chamar uma ou mais funções para auxiliar na consulta do usuário.\n\nVocê recebe assinaturas de funções dentro de tags XML <tools></tools>:\n<tools>" }}
{%- for tool in tools %}
{{- "\n" }}
{{- tool | tojson }}
{%- endfor %}
{{- "\n</tools>\n\nPara cada chamada de função, retorne um objeto json com o nome da função e os argumentos dentro das tags XML <tool_call></tool_call>:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
{%- else %}
{#- Standard system message without tools #}
{%- if messages[0].role == 'system' %}
{{- '<|im_start|>system\n' + messages[0].content + '<|im_end|>\n' }}
{%- endif %}
{%- endif %}
{#- Detect multi-step tool usage by finding the last real user query #}
{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
{%- for message in messages[::-1] %}
{%- set index = (messages|length - 1) - loop.index0 %}
{%- if ns.multi_step_tool and message.role == "user" and message.content is string and not(message.content.startswith('<tool_response>') and message.content.endswith('</tool_response>')) %}
{%- set ns.multi_step_tool = false %}
{%- set ns.last_query_index = index %}
{%- endif %}
{%- endfor %}
{#- Process each message in the conversation #}
{%- for message in messages %}
{#- Normalize content to string #}
{%- if message.content is string %}
{%- set content = message.content %}
{%- else %}
{%- set content = '' %}
{%- endif %}
{#- Handle user messages and non-first system messages #}
{%- if (message.role == "user") or (message.role == "system" and not loop.first) %}
{{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
{#- Handle assistant messages with reasoning #}
{%- elif message.role == "assistant" %}
{#- Extract reasoning content if present #}
{%- set reasoning_content = '' %}
{%- if message.reasoning_content is string %}
{%- set reasoning_content = message.reasoning_content %}
{%- else %}
{#- Parse <think></think> tags from content #}
{%- if '</think>' in content %}
{%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
{%- set content = content.split('</think>')[-1].lstrip('\n') %}
{%- endif %}
{%- endif %}
{{- '<|im_start|>' + message.role }}
{% generation %}
{#- Add reasoning tags for messages after last user query #}
{%- if loop.index0 > ns.last_query_index %}
{%- if loop.last or (not loop.last and reasoning_content) %}
{{- '<think>\n' + reasoning_content.strip('\n') + '\n</think>\n\n' + content.lstrip('\n') }}
{%- else %}
{{- content }}
{%- endif %}
{%- else %}
{{- content }}
{%- endif %}
{#- Add tool calls if present #}
{%- if message.tool_calls %}
{%- for tool_call in message.tool_calls %}
{%- if (loop.first and content) or (not loop.first) %}
{{- '\n' }}
{%- endif %}
{#- Normalize tool call format #}
{%- if tool_call.function %}
{%- set tool_call = tool_call.function %}
{%- endif %}
{{- '<tool_call>\n{"name": "' }}
{{- tool_call.name }}
{{- '", "arguments": ' }}
{%- if tool_call.arguments is string %}
{{- tool_call.arguments }}
{%- else %}
{{- tool_call.arguments | tojson }}
{%- endif %}
{{- '}\n</tool_call>' }}
{%- endfor %}
{%- endif %}
{{- '<|im_end|>' }}
{% endgeneration %}
{#- Handle tool response messages #}
{%- elif message.role == "tool" %}
{#- Group consecutive tool responses under one user message #}
{%- if loop.first or (messages[loop.index0 - 1].role != "tool") %}
{{- '<|im_start|>user' }}
{%- endif %}
{{- '\n<tool_response>\n' }}
{{- content }}
{{- '\n</tool_response>' }}
{%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
{{- '<|im_end|>\n' }}
{%- endif %}
{%- endif %}
{%- endfor %}
{#- Add generation prompt if requested #}
{%- if add_generation_prompt %}
{{- '<|im_start|>assistant\n' }}
{%- endif %}

61
config.json Normal file
View File

@@ -0,0 +1,61 @@
{
"architectures": [
"Qwen3ForCausalLM"
],
"attention_bias": false,
"attention_dropout": 0.0,
"bos_token_id": 1,
"dtype": "bfloat16",
"eos_token_id": 2,
"head_dim": 128,
"hidden_act": "silu",
"hidden_size": 2048,
"initializer_range": 0.02,
"intermediate_size": 6144,
"layer_types": [
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention"
],
"max_position_embeddings": 4096,
"max_window_layers": 28,
"model_type": "qwen3",
"num_attention_heads": 16,
"num_hidden_layers": 28,
"num_key_value_heads": 8,
"pad_token_id": 49109,
"rms_norm_eps": 1e-06,
"rope_scaling": null,
"rope_theta": 1000000,
"sliding_window": null,
"tie_word_embeddings": true,
"transformers_version": "4.57.3",
"use_cache": false,
"use_sliding_window": false,
"vocab_size": 49152
}

207
evals.yaml Normal file
View File

@@ -0,0 +1,207 @@
evaluations:
arc_challenge_poly_pt_acc: 0.37948717948717947
arc_challenge_poly_pt_acc_norm: 0.4282051282051282
arc_challenge_poly_pt_acc_norm_stderr: 0.014472341586181985
arc_challenge_poly_pt_acc_stderr: 0.014192754090886793
arc_challenge_poly_pt_alias: arc_challenge_poly_pt
assin2_rte_acc,all: 0.8705065359477124
assin2_rte_acc_stderr,all: 0.004798540986291823
assin2_rte_alias: assin2_rte
assin2_rte_f1_macro,all: 0.8691746625542952
assin2_rte_f1_macro_stderr,all: 0.004847278802663799
assin2_sts_alias: assin2_sts
assin2_sts_mse,all: 2.2781209150326798
assin2_sts_mse_stderr,all: N/A
assin2_sts_pearson,all: 0.47716809267162485
assin2_sts_pearson_stderr,all: 0.014407170423859306
assin_entailment_acc: 0.7325
assin_entailment_acc_stderr: 0.006999870502142285
assin_entailment_alias: assin_entailment
assin_paraphrase_acc: 0.73625
assin_paraphrase_acc_stderr: 0.006968401827607802
assin_paraphrase_alias: assin_paraphrase
belebele_por_Latn_acc: 0.6766666666666666
belebele_por_Latn_acc_norm: 0.6766666666666666
belebele_por_Latn_acc_norm_stderr: 0.015600294087844632
belebele_por_Latn_acc_stderr: 0.015600294087844632
belebele_por_Latn_alias: belebele_por_Latn
bluex_acc,all: 0.39221140472878996
bluex_acc,exam_id__UNICAMP_2018: 0.48148148148148145
bluex_acc,exam_id__UNICAMP_2019: 0.42
bluex_acc,exam_id__UNICAMP_2020: 0.41818181818181815
bluex_acc,exam_id__UNICAMP_2021_1: 0.4782608695652174
bluex_acc,exam_id__UNICAMP_2021_2: 0.29411764705882354
bluex_acc,exam_id__UNICAMP_2022: 0.5384615384615384
bluex_acc,exam_id__UNICAMP_2023: 0.4883720930232558
bluex_acc,exam_id__UNICAMP_2024: 0.4444444444444444
bluex_acc,exam_id__USP_2018: 0.24074074074074073
bluex_acc,exam_id__USP_2019: 0.35
bluex_acc,exam_id__USP_2020: 0.39285714285714285
bluex_acc,exam_id__USP_2021: 0.3076923076923077
bluex_acc,exam_id__USP_2022: 0.24489795918367346
bluex_acc,exam_id__USP_2023: 0.38636363636363635
bluex_acc,exam_id__USP_2024: 0.4634146341463415
bluex_acc_stderr,all: 0.010527552369043944
bluex_acc_stderr,exam_id__UNICAMP_2018: 0.03932520147192548
bluex_acc_stderr,exam_id__UNICAMP_2019: 0.040409592319981036
bluex_acc_stderr,exam_id__UNICAMP_2020: 0.038500996629012005
bluex_acc_stderr,exam_id__UNICAMP_2021_1: 0.04265317780472471
bluex_acc_stderr,exam_id__UNICAMP_2021_2: 0.0369309745845929
bluex_acc_stderr,exam_id__UNICAMP_2022: 0.04616267465694302
bluex_acc_stderr,exam_id__UNICAMP_2023: 0.04391771882582978
bluex_acc_stderr,exam_id__UNICAMP_2024: 0.04266417007528182
bluex_acc_stderr,exam_id__USP_2018: 0.03361325629183327
bluex_acc_stderr,exam_id__USP_2019: 0.04344929186946837
bluex_acc_stderr,exam_id__USP_2020: 0.037457169801895486
bluex_acc_stderr,exam_id__USP_2021: 0.03696695228727266
bluex_acc_stderr,exam_id__USP_2022: 0.03545185428690165
bluex_acc_stderr,exam_id__USP_2023: 0.042474562735645566
bluex_acc_stderr,exam_id__USP_2024: 0.04499354148452882
bluex_alias: bluex
calame_pt_acc: 0.11127167630057803
calame_pt_acc_stderr: 0.00690347530269077
calame_pt_alias: calame_pt
calame_pt_perplexity: 5965.407764074068
calame_pt_perplexity_stderr: 683.2371253201818
enem_challenge_acc,all: 0.3988803358992302
enem_challenge_acc,exam_id__2009: 0.46956521739130436
enem_challenge_acc,exam_id__2010: 0.5128205128205128
enem_challenge_acc,exam_id__2011: 0.46153846153846156
enem_challenge_acc,exam_id__2012: 0.4224137931034483
enem_challenge_acc,exam_id__2013: 0.3148148148148148
enem_challenge_acc,exam_id__2014: 0.41284403669724773
enem_challenge_acc,exam_id__2015: 0.40336134453781514
enem_challenge_acc,exam_id__2016: 0.36363636363636365
enem_challenge_acc,exam_id__2016_2: 0.4065040650406504
enem_challenge_acc,exam_id__2017: 0.39655172413793105
enem_challenge_acc,exam_id__2022: 0.3157894736842105
enem_challenge_acc,exam_id__2023: 0.32592592592592595
enem_challenge_acc_stderr,all: 0.00749197527112301
enem_challenge_acc_stderr,exam_id__2009: 0.02676517541031027
enem_challenge_acc_stderr,exam_id__2010: 0.02676561702634192
enem_challenge_acc_stderr,exam_id__2011: 0.026575632846671127
enem_challenge_acc_stderr,exam_id__2012: 0.02652807977680181
enem_challenge_acc_stderr,exam_id__2013: 0.02570106484445073
enem_challenge_acc_stderr,exam_id__2014: 0.02722456002129014
enem_challenge_acc_stderr,exam_id__2015: 0.025882945505526844
enem_challenge_acc_stderr,exam_id__2016: 0.025343746567122242
enem_challenge_acc_stderr,exam_id__2016_2: 0.025491304109387235
enem_challenge_acc_stderr,exam_id__2017: 0.0261642629559971
enem_challenge_acc_stderr,exam_id__2022: 0.02331831926729694
enem_challenge_acc_stderr,exam_id__2023: 0.023372974631758865
enem_challenge_alias: enem
faquad_nli_acc,all: 0.7892307692307692
faquad_nli_acc_stderr,all: 0.011279066985135663
faquad_nli_alias: faquad_nli
faquad_nli_f1_macro,all: 0.49256657036543183
faquad_nli_f1_macro_stderr,all: 0.01318624622535848
global_piqa_completions_por_latn_braz_acc: 0.75
global_piqa_completions_por_latn_braz_acc_bytes: 0.74
global_piqa_completions_por_latn_braz_acc_bytes_stderr: 0.0440844002276808
global_piqa_completions_por_latn_braz_acc_norm: 0.74
global_piqa_completions_por_latn_braz_acc_norm_stderr: 0.0440844002276808
global_piqa_completions_por_latn_braz_acc_stderr: 0.04351941398892446
global_piqa_completions_por_latn_braz_alias: global_piqa_completions_por_latn_braz
gsm8k_pt_alias: gsm8k_pt
gsm8k_pt_exact_match,flexible-extract: 0.228310502283105
gsm8k_pt_exact_match,strict-match: 0.0
gsm8k_pt_exact_match_stderr,flexible-extract: 0.011583822031121497
gsm8k_pt_exact_match_stderr,strict-match: 0.0
hatebr_offensive_acc,all: 0.7071428571428572
hatebr_offensive_acc_stderr,all: 0.008611736471385361
hatebr_offensive_alias: hatebr_offensive_binary
hatebr_offensive_f1_macro,all: 0.7019751844399732
hatebr_offensive_f1_macro_stderr,all: 0.008866946580220815
hellaswag_poly_pt_acc: 0.4334164048109221
hellaswag_poly_pt_acc_norm: 0.5494636471990465
hellaswag_poly_pt_acc_norm_stderr: 0.005179413791359488
hellaswag_poly_pt_acc_stderr: 0.005158588405358706
hellaswag_poly_pt_alias: hellaswag_poly_pt
humaneval_instruct_alias: humaneval_instruct
humaneval_instruct_pass@1,create_test: 0.0
humaneval_instruct_pass@1_stderr,create_test: 0.0
ifeval_pt_alias: ifeval_pt
ifeval_pt_inst_level_loose_acc: 0.4511627906976744
ifeval_pt_inst_level_loose_acc_stderr: N/A
ifeval_pt_inst_level_strict_acc: 0.3395348837209302
ifeval_pt_inst_level_strict_acc_stderr: N/A
ifeval_pt_prompt_level_loose_acc: 0.33666666666666667
ifeval_pt_prompt_level_loose_acc_stderr: 0.02732941756218688
ifeval_pt_prompt_level_strict_acc: 0.23333333333333334
ifeval_pt_prompt_level_strict_acc_stderr: 0.024459979523511425
lambada_poly_pt_acc: 0.267028915195032
lambada_poly_pt_acc_stderr: 0.00616360274242743
lambada_poly_pt_alias: lambada_poly_pt
lambada_poly_pt_perplexity: 214.12248648482947
lambada_poly_pt_perplexity_stderr: 13.012850955803207
mmlu_poly_pt_acc: 0.43297808465926146
mmlu_poly_pt_acc_stderr: 0.004292713120965234
mmlu_poly_pt_alias: mmlu_poly_pt
oab_exams_acc,all: 0.3425968109339408
oab_exams_acc,exam_id__2010-01: 0.3176470588235294
oab_exams_acc,exam_id__2010-02: 0.33
oab_exams_acc,exam_id__2011-03: 0.3333333333333333
oab_exams_acc,exam_id__2011-04: 0.3625
oab_exams_acc,exam_id__2011-05: 0.325
oab_exams_acc,exam_id__2012-06: 0.35
oab_exams_acc,exam_id__2012-06a: 0.325
oab_exams_acc,exam_id__2012-07: 0.3375
oab_exams_acc,exam_id__2012-08: 0.325
oab_exams_acc,exam_id__2012-09: 0.24675324675324675
oab_exams_acc,exam_id__2013-10: 0.325
oab_exams_acc,exam_id__2013-11: 0.2875
oab_exams_acc,exam_id__2013-12: 0.3375
oab_exams_acc,exam_id__2014-13: 0.3875
oab_exams_acc,exam_id__2014-14: 0.325
oab_exams_acc,exam_id__2014-15: 0.41025641025641024
oab_exams_acc,exam_id__2015-16: 0.3125
oab_exams_acc,exam_id__2015-17: 0.3974358974358974
oab_exams_acc,exam_id__2015-18: 0.375
oab_exams_acc,exam_id__2016-19: 0.41025641025641024
oab_exams_acc,exam_id__2016-20: 0.4375
oab_exams_acc,exam_id__2016-20a: 0.325
oab_exams_acc,exam_id__2016-21: 0.25
oab_exams_acc,exam_id__2017-22: 0.375
oab_exams_acc,exam_id__2017-23: 0.3125
oab_exams_acc,exam_id__2017-24: 0.4
oab_exams_acc,exam_id__2018-25: 0.3375
oab_exams_acc_stderr,all: 0.005854656489923565
oab_exams_acc_stderr,exam_id__2010-01: 0.02913844109530932
oab_exams_acc_stderr,exam_id__2010-02: 0.02720413223389449
oab_exams_acc_stderr,exam_id__2011-03: 0.02739274623256926
oab_exams_acc_stderr,exam_id__2011-04: 0.03103610079124539
oab_exams_acc_stderr,exam_id__2011-05: 0.030180181541041882
oab_exams_acc_stderr,exam_id__2012-06: 0.030663426761582978
oab_exams_acc_stderr,exam_id__2012-06a: 0.03019542051107743
oab_exams_acc_stderr,exam_id__2012-07: 0.030587368127940558
oab_exams_acc_stderr,exam_id__2012-08: 0.030150394921254077
oab_exams_acc_stderr,exam_id__2012-09: 0.028367169852259254
oab_exams_acc_stderr,exam_id__2013-10: 0.030312606625205758
oab_exams_acc_stderr,exam_id__2013-11: 0.0291233179987808
oab_exams_acc_stderr,exam_id__2013-12: 0.030508999749351583
oab_exams_acc_stderr,exam_id__2014-13: 0.03131790740981766
oab_exams_acc_stderr,exam_id__2014-14: 0.030290399093110758
oab_exams_acc_stderr,exam_id__2014-15: 0.032059169494021815
oab_exams_acc_stderr,exam_id__2015-16: 0.029833251327948757
oab_exams_acc_stderr,exam_id__2015-17: 0.03201117621919014
oab_exams_acc_stderr,exam_id__2015-18: 0.0311783808239491
oab_exams_acc_stderr,exam_id__2016-19: 0.032162738371024235
oab_exams_acc_stderr,exam_id__2016-20: 0.031934266796334064
oab_exams_acc_stderr,exam_id__2016-20a: 0.030215012978448332
oab_exams_acc_stderr,exam_id__2016-21: 0.027979645441821365
oab_exams_acc_stderr,exam_id__2017-22: 0.031096010380049006
oab_exams_acc_stderr,exam_id__2017-23: 0.02995046656597778
oab_exams_acc_stderr,exam_id__2017-24: 0.03158083933896483
oab_exams_acc_stderr,exam_id__2018-25: 0.030481185333792096
oab_exams_alias: oab_exams
portuguese_hate_speech_acc,all: 0.7097532314923619
portuguese_hate_speech_acc_stderr,all: 0.01097429657035187
portuguese_hate_speech_alias: portuguese_hate_speech_binary
portuguese_hate_speech_f1_macro,all: 0.6922452504317789
portuguese_hate_speech_f1_macro_stderr,all: 0.011709050058396507
tweetsentbr_acc,all: 0.6422885572139303
tweetsentbr_acc_stderr,all: 0.007563312856378464
tweetsentbr_alias: tweetsentbr
tweetsentbr_f1_macro,all: 0.6428335432993848
tweetsentbr_f1_macro_stderr,all: 0.007483756787178434
step: 3595

14
generation_config.json Normal file
View File

@@ -0,0 +1,14 @@
{
"bos_token_id": 1,
"do_sample": true,
"eos_token_id": [
2
],
"max_new_tokens": 1024,
"pad_token_id": 49109,
"renormalize_logits": true,
"repetition_penalty": 1.2,
"temperature": 0.1,
"transformers_version": "4.57.3",
"use_cache": false
}

3
logo.png Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:1856d91c3b35390cee5122902d94044657c67df7034ca4005316275c404fc8a0
size 197189

3
model.safetensors Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:444dac3b9e519030276ff7d59d86e392bfb021e6e589f0ebab12e4fff3b25a9b
size 3020182248

82
ruler.yaml Normal file
View File

@@ -0,0 +1,82 @@
model_name: Tucano2-qwen-1.5B-Think
results:
niah_pt_multikey_1_1024: 0.706
niah_pt_multikey_1_1024_stderr: 0.02039509548493655
niah_pt_multikey_1_2048: 0.67
niah_pt_multikey_1_2048_stderr: 0.021049612166134782
niah_pt_multikey_1_4096: 0.542
niah_pt_multikey_1_4096_stderr: N/A
niah_pt_multikey_1_alias: " - niah_pt_multikey_1"
niah_pt_multikey_2_1024: 0.48
niah_pt_multikey_2_1024_stderr: 0.022365160424231326
niah_pt_multikey_2_2048: 0.238
niah_pt_multikey_2_2048_stderr: 0.019064072958198387
niah_pt_multikey_2_4096: 0.062
niah_pt_multikey_2_4096_stderr: N/A
niah_pt_multikey_2_alias: " - niah_pt_multikey_2"
niah_pt_multikey_3_1024: 0.486
niah_pt_multikey_3_1024_stderr: 0.022374298166353144
niah_pt_multikey_3_2048: 0.31
niah_pt_multikey_3_2048_stderr: 0.020704041021724684
niah_pt_multikey_3_4096: 0.184
niah_pt_multikey_3_4096_stderr: N/A
niah_pt_multikey_3_alias: " - niah_pt_multikey_3"
niah_pt_multiquery_1024: 0.531
niah_pt_multiquery_1024_stderr: 0.013691344193015646
niah_pt_multiquery_2048: 0.4915
niah_pt_multiquery_2048_stderr: 0.014053487147395266
niah_pt_multiquery_4096: 0.4215
niah_pt_multiquery_4096_stderr: N/A
niah_pt_multiquery_alias: " - niah_pt_multiquery"
niah_pt_multivalue_1024: 0.4995
niah_pt_multivalue_1024_stderr: 0.013476376569794338
niah_pt_multivalue_2048: 0.519
niah_pt_multivalue_2048_stderr: 0.013327913059930505
niah_pt_multivalue_4096: 0.4545
niah_pt_multivalue_4096_stderr: N/A
niah_pt_multivalue_alias: " - niah_pt_multivalue"
niah_pt_single_1_1024: 0.82
niah_pt_single_1_1024_stderr: 0.017198592476314233
niah_pt_single_1_2048: 0.816
niah_pt_single_1_2048_stderr: 0.017346174781752842
niah_pt_single_1_4096: 0.8
niah_pt_single_1_4096_stderr: N/A
niah_pt_single_1_alias: " - niah_pt_single_1"
niah_pt_single_2_1024: 0.778
niah_pt_single_2_1024_stderr: 0.018604414758250098
niah_pt_single_2_2048: 0.772
niah_pt_single_2_2048_stderr: 0.018781306529363172
niah_pt_single_2_4096: 0.688
niah_pt_single_2_4096_stderr: N/A
niah_pt_single_2_alias: " - niah_pt_single_2"
niah_pt_single_3_1024: 0.468
niah_pt_single_3_1024_stderr: 0.022337186479044296
niah_pt_single_3_2048: 0.508
niah_pt_single_3_2048_stderr: 0.022380208834928014
niah_pt_single_3_4096: 0.5
niah_pt_single_3_4096_stderr: N/A
niah_pt_single_3_alias: " - niah_pt_single_3"
ruler_pt_4096: 0.44008484848484847
ruler_pt_4096_stderr: N/A
ruler_pt_alias: ruler_pt
ruler_pt_cwe_1024: 0.2516
ruler_pt_cwe_1024_stderr: 0.0065879953982022075
ruler_pt_cwe_2048: 0.10560000000000001
ruler_pt_cwe_2048_stderr: 0.0046626989526502875
ruler_pt_cwe_4096: 0.268
ruler_pt_cwe_4096_stderr: N/A
ruler_pt_cwe_alias: " - ruler_pt_cwe"
ruler_pt_fwe_1024: 0.7766666666666666
ruler_pt_fwe_1024_stderr: 0.010771818051204566
ruler_pt_fwe_2048: 0.644
ruler_pt_fwe_2048_stderr: 0.010544896116732008
ruler_pt_fwe_4096: 0.5413333333333332
ruler_pt_fwe_4096_stderr: N/A
ruler_pt_fwe_alias: " - ruler_pt_fwe"
ruler_pt_vt_1024: 0.8336
ruler_pt_vt_1024_stderr: 0.01194434656352784
ruler_pt_vt_2048: 0.4344
ruler_pt_vt_2048_stderr: 0.014185758756689964
ruler_pt_vt_4096: 0.37960000000000005
ruler_pt_vt_4096_stderr: N/A
ruler_pt_vt_alias: " - ruler_pt_vt"

30
special_tokens_map.json Normal file
View File

@@ -0,0 +1,30 @@
{
"bos_token": {
"content": "<|im_start|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false
},
"eos_token": {
"content": "<|im_end|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false
},
"pad_token": {
"content": "<|pad|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false
},
"unk_token": {
"content": "<|unk|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false
}
}

463711
tokenizer.json Normal file

File diff suppressed because it is too large Load Diff

397
tokenizer_config.json Normal file
View File

@@ -0,0 +1,397 @@
{
"add_bos_token": false,
"add_eos_token": false,
"add_prefix_space": null,
"added_tokens_decoder": {
"0": {
"content": "<|unk|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"1": {
"content": "<|im_start|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"2": {
"content": "<|im_end|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"49109": {
"content": "<|pad|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"49110": {
"content": "<tools>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49111": {
"content": "</tools>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49112": {
"content": "<tool_call>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49113": {
"content": "</tool_call>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49114": {
"content": "<tool_response>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49115": {
"content": "</tool_response>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49116": {
"content": "<think>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49117": {
"content": "</think>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49118": {
"content": "<answer>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49119": {
"content": "</answer>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49120": {
"content": "<context>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49121": {
"content": "</context>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49122": {
"content": "<|fim_prefix|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49123": {
"content": "<|fim_suffix|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49124": {
"content": "<|fim_middle|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49125": {
"content": "<|repo_name|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49126": {
"content": "<|image|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49127": {
"content": "<|image_pad|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49128": {
"content": "<|image_placeholder|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49129": {
"content": " ",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49130": {
"content": " ",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49131": {
"content": " ",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49132": {
"content": " ",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49133": {
"content": " ",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49134": {
"content": " ",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49135": {
"content": " ",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49136": {
"content": " ",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49137": {
"content": " ",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49138": {
"content": " ",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49139": {
"content": " ",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49140": {
"content": " ",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49141": {
"content": " ",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49142": {
"content": " ",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49143": {
"content": " ",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49144": {
"content": " ",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49145": {
"content": " ",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49146": {
"content": " ",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49147": {
"content": " ",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49148": {
"content": " ",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"49149": {
"content": " ",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49150": {
"content": " ",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"49151": {
"content": " ",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
}
},
"bos_token": "<|im_start|>",
"bos_token_id": 1,
"clean_up_tokenization_spaces": false,
"eos_token": "<|im_end|>",
"eos_token_id": 2,
"extra_special_tokens": {},
"legacy": false,
"model_input_names": [
"input_ids",
"attention_mask"
],
"model_max_length": 4096,
"pad_token": "<|pad|>",
"pad_token_id": 49109,
"padding_side": "right",
"sp_model_kwargs": {},
"spaces_between_special_tokens": false,
"tokenizer_class": "PreTrainedTokenizerFast",
"truncation_side": "right",
"unk_token": "<|unk|>",
"unk_token_id": 0,
"use_default_system_prompt": false
}

3
train_logs_apo.parquet Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:17955b2e6b87674c99c03a62da0015c4109d0bdb29dff4c0ed0d7c72897ba58e
size 46633

3
train_logs_sft.parquet Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:e96a3c5634dcc1082cefe019f9e271da5e9ca6a9c7cf02823d48f18f8f44baa7
size 66628

98
training_config_apo.yaml Normal file
View File

@@ -0,0 +1,98 @@
# Directory settings
checkpoint_dir: "/polyglot/portuguese/checkpoints/models/Tucano2-qwen-1.5B-Think"
train_dataset_dir:
# Total: 13,649 samples (x5 epochs)
# Harmfull samples (with reasoning): 4,008 samples
- /polyglot/portuguese/gigaverbo-v2-dpo/harmfull-reasoning
# Harmless samples (with reasoning): 9,641 samples
- /polyglot/portuguese/gigaverbo-v2-dpo/harmless-reasoning
val_dataset_dir: null
dataset_type: "jsonl"
cache_dir: "/lustre/mlnvme/data/polyglot/.cache"
# Data loading settings
pin_memory: true
num_workers_for_dataloader: 16
shuffle_dataset: true
mask_eos_token: false
mask_pad_token: false
# Model architecture settings
vocab_size: 49152
num_hidden_layers: 28
num_attention_heads: 16
num_key_value_heads: 8
head_dim: 128
hidden_size: 2048
intermediate_size: 6144
max_position_embeddings: 4096
tie_word_embeddings: true
hidden_act: "silu"
output_hidden_states: false
attn_implementation: "flash_attention_2"
use_cache: false
no_rope_layer_interval: null
rope_theta: 1000000.0
rope_scale_factor: null
rms_norm_eps: 0.000001
# Training settings
total_batch_size: 524288
micro_batch_size: 4
gradient_accumulation_steps: 4
eval_micro_batch_size: null
num_train_epochs: 5
warmup_ratio: 0.1
max_learning_rate: 0.000005
min_learning_rate: 0.0
muon_learning_rate: null
weight_decay: 0.0
beta1: 0.9
beta2: 0.95
eps: 0.00000001
lr_decay_type: "cosine"
use_sqrt: false
lr_decay_iters_coef: 1.
seed: 42
max_steps: 535
max_grad_norm: 1.0
# APO settings
loss_type: "apo_zero"
dpo_beta: 0.5
precompute_ref_log_probs: true
truncation_mode: "keep_end"
# Precision and optimization settings
torch_compile: false
mat_mul_precision: "highest"
tf32: true
bf16: true
gradient_checkpointing: true
use_liger_kernel: false
static_graph: false
# Hub settings
push_to_hub: false
hub_token: null
hub_model_id: null
# Tokenizer and Reference model
tokenizer_name_or_path: "/polyglot/portuguese/checkpoints/models/Tucano2-qwen-1.5B-Think-SFT"
chat_template_path: null
reference_model: "/polyglot/portuguese/checkpoints/models/Tucano2-qwen-1.5B-Think-SFT"
continual_pretraining: true
# Checkpoint settings
resume_from_checkpoint: null
checkpointing_steps: 1000
begin_new_stage: true
stage_name: "single_cosine"
# Miscellaneous settings
sanity_check: false
sanity_check_num_samples: 100000
wandb_token: null
wandb_id: "tucano2-qwen-1.5b-think-apo"
wandb_project: "Polyglot"
wandb_desc: "Developing LLMs for low-resource languages"

93
training_config_sft.yaml Normal file
View File

@@ -0,0 +1,93 @@
# Directory settings
checkpoint_dir: "/polyglot/portuguese/checkpoints/models/Tucano2-qwen-1.5B-Think-SFT"
train_dataset_dir:
# Reasoning: ~34 million tokens (x5 epochs)
- /polyglot/portuguese/gigaverbo-v2-sft/reasoning
val_dataset_dir: null
dataset_type: "jsonl"
cache_dir: "/lustre/mlnvme/data/polyglot/.cache"
# Data loading settings
pin_memory: true
num_workers_for_dataloader: 16
shuffle_dataset: true
mask_eos_token: false
mask_pad_token: true
# Model architecture settings
vocab_size: 49152
num_hidden_layers: 28
num_attention_heads: 16
num_key_value_heads: 8
head_dim: 128
hidden_size: 2048
intermediate_size: 6144
max_position_embeddings: 4096
tie_word_embeddings: true
hidden_act: "silu"
output_hidden_states: false
attn_implementation: "flash_attention_2"
use_cache: false
no_rope_layer_interval: null
rope_theta: 1000000.0
rope_scale_factor: null
rms_norm_eps: 0.000001
# Training settings
total_batch_size: 524288
micro_batch_size: 4
gradient_accumulation_steps: 4
eval_micro_batch_size: null
num_train_epochs: 5
warmup_ratio: 0.1
max_learning_rate: 0.000075
min_learning_rate: 0.0
muon_learning_rate: null
weight_decay: 0.0
beta1: 0.9
beta2: 0.95
eps: 0.00000001
lr_decay_type: "cosine"
use_sqrt: false
lr_decay_iters_coef: 1.
seed: 42
max_steps: 3060
max_grad_norm: 1.0
# SFT settings
packing: false
assistant_only_loss: true
# Precision and optimization settings
torch_compile: false
mat_mul_precision: "highest"
tf32: true
bf16: true
gradient_checkpointing: true
use_liger_kernel: true
static_graph: false
# Hub settings
push_to_hub: false
hub_token: null
hub_model_id: null
# Tokenizer and Reference model
tokenizer_name_or_path: "Polygl0t/Tucano2-qwen-1.5B-Base"
chat_template_path: null
reference_model: "Polygl0t/Tucano2-qwen-1.5B-Base"
continual_pretraining: true
# Checkpoint settings
resume_from_checkpoint: null
checkpointing_steps: 1000
begin_new_stage: true
stage_name: "single_cosine"
# Miscellaneous settings
sanity_check: false
sanity_check_num_samples: 100000
wandb_token: null
wandb_id: "tucano2-qwen-1.5b-think-sft"
wandb_project: "Polyglot"
wandb_desc: "Developing LLMs for low-resource languages"