初始化项目,由ModelHub XC社区提供模型
Model: fgoose180/zagreus-0.4B-xmoons-improved Source: Original Platform
This commit is contained in:
37
.gitattributes
vendored
Normal file
37
.gitattributes
vendored
Normal file
@@ -0,0 +1,37 @@
|
||||
*.7z filter=lfs diff=lfs merge=lfs -text
|
||||
*.arrow filter=lfs diff=lfs merge=lfs -text
|
||||
*.bin filter=lfs diff=lfs merge=lfs -text
|
||||
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
||||
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
||||
*.ftz filter=lfs diff=lfs merge=lfs -text
|
||||
*.gz filter=lfs diff=lfs merge=lfs -text
|
||||
*.h5 filter=lfs diff=lfs merge=lfs -text
|
||||
*.joblib filter=lfs diff=lfs merge=lfs -text
|
||||
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
||||
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
||||
*.model filter=lfs diff=lfs merge=lfs -text
|
||||
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
||||
*.npy filter=lfs diff=lfs merge=lfs -text
|
||||
*.npz filter=lfs diff=lfs merge=lfs -text
|
||||
*.onnx filter=lfs diff=lfs merge=lfs -text
|
||||
*.ot filter=lfs diff=lfs merge=lfs -text
|
||||
*.parquet filter=lfs diff=lfs merge=lfs -text
|
||||
*.pb filter=lfs diff=lfs merge=lfs -text
|
||||
*.pickle filter=lfs diff=lfs merge=lfs -text
|
||||
*.pkl filter=lfs diff=lfs merge=lfs -text
|
||||
*.pt filter=lfs diff=lfs merge=lfs -text
|
||||
*.pth filter=lfs diff=lfs merge=lfs -text
|
||||
*.rar filter=lfs diff=lfs merge=lfs -text
|
||||
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
||||
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
||||
*.tar filter=lfs diff=lfs merge=lfs -text
|
||||
*.tflite filter=lfs diff=lfs merge=lfs -text
|
||||
*.tgz filter=lfs diff=lfs merge=lfs -text
|
||||
*.wasm filter=lfs diff=lfs merge=lfs -text
|
||||
*.xz filter=lfs diff=lfs merge=lfs -text
|
||||
*.zip filter=lfs diff=lfs merge=lfs -text
|
||||
*.zst filter=lfs diff=lfs merge=lfs -text
|
||||
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
||||
tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
||||
zagreus-final/tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
||||
242
README.md
Normal file
242
README.md
Normal file
@@ -0,0 +1,242 @@
|
||||
---
|
||||
library_name: transformers
|
||||
pipeline_tag: text-generation
|
||||
language:
|
||||
- it
|
||||
license: llama3
|
||||
base_model:
|
||||
- mii-llm/zagreus-0.4B-ita
|
||||
datasets:
|
||||
- efederici/pinocchio
|
||||
- FinancialSupport/italic_sft
|
||||
- FinancialSupport/italic_sft_ext
|
||||
- FinancialSupport/quiz_militare
|
||||
tags:
|
||||
- llama
|
||||
- italian
|
||||
- small-language-model
|
||||
- multiple-choice
|
||||
- supervised-fine-tuning
|
||||
- post-training
|
||||
- italic
|
||||
---
|
||||
|
||||
# Zagreus 0.4B xmoons improved
|
||||
|
||||
Zagreus 0.4B xmoons improved is an Italian multiple-choice model fine-tuned
|
||||
from [`mii-llm/zagreus-0.4B-ita`](https://huggingface.co/mii-llm/zagreus-0.4B-ita).
|
||||
It is the selected checkpoint from the reproducible H100 run that reached
|
||||
**44.35% accuracy** on the official 10,000-question ITALIC benchmark in fast,
|
||||
five-shot mode through vLLM.
|
||||
|
||||
The model was developed for the
|
||||
[`mii-llm/Post-Training-Challenge`](https://huggingface.co/spaces/mii-llm/Post-Training-Challenge).
|
||||
|
||||
The model is a compact Llama-style causal language model with 437,760,960
|
||||
parameters. Its tokenizer and Llama 3 chat template come from
|
||||
[`swap-uniba/LLaMAntino-3-ANITA-8B-Inst-DPO-ITA`](https://huggingface.co/swap-uniba/LLaMAntino-3-ANITA-8B-Inst-DPO-ITA).
|
||||
The repository is a self-contained Transformers checkpoint and uses
|
||||
`<|eot_id|>` as both the end-of-turn and generation stop token.
|
||||
|
||||
## Intended use
|
||||
|
||||
The model is intended for:
|
||||
|
||||
- research on compact Italian language models;
|
||||
- Italian multiple-choice question answering;
|
||||
- participation in the MII Post-Training Challenge;
|
||||
- reproduction and analysis of the ITALIC post-training experiment.
|
||||
|
||||
It is not designed as a general-purpose assistant, a factual authority, or a
|
||||
component for medical, legal, financial, or other high-stakes decisions.
|
||||
|
||||
## Usage
|
||||
|
||||
Install PyTorch and Transformers, then load the checkpoint from the repository
|
||||
root:
|
||||
|
||||
```python
|
||||
import torch
|
||||
from transformers import AutoModelForCausalLM, AutoTokenizer
|
||||
|
||||
model_id = "fgoose180/zagreus-0.4B-xmoons-improved"
|
||||
device = "cuda" if torch.cuda.is_available() else "cpu"
|
||||
dtype = torch.bfloat16 if device == "cuda" else torch.float32
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained(model_id)
|
||||
model = AutoModelForCausalLM.from_pretrained(
|
||||
model_id,
|
||||
dtype=dtype,
|
||||
).to(device)
|
||||
|
||||
messages = [
|
||||
{"role": "system", "content": "Sei un assistente utile."},
|
||||
{
|
||||
"role": "user",
|
||||
"content": """Rispondi alla seguente domanda a scelta multipla sull'argomento 'data_geography'. La tua risposta deve essere nel seguente formato: 'LETTERA' (senza virgolette) dove LETTERA è una tra ABCD. Scrivi solo la lettera corrispondente alla tua risposta senza spiegazioni.
|
||||
|
||||
Qual è la capitale d'Italia?
|
||||
|
||||
A) Roma
|
||||
B) Milano
|
||||
C) Torino
|
||||
D) Napoli
|
||||
|
||||
Risposta:""",
|
||||
},
|
||||
]
|
||||
|
||||
inputs = tokenizer.apply_chat_template(
|
||||
messages,
|
||||
add_generation_prompt=True,
|
||||
return_tensors="pt",
|
||||
return_dict=True,
|
||||
).to(device)
|
||||
|
||||
with torch.inference_mode():
|
||||
output = model.generate(
|
||||
**inputs,
|
||||
max_new_tokens=8,
|
||||
do_sample=False,
|
||||
eos_token_id=tokenizer.eos_token_id,
|
||||
pad_token_id=tokenizer.pad_token_id,
|
||||
)
|
||||
|
||||
answer = tokenizer.decode(
|
||||
output[0, inputs["input_ids"].shape[1] :],
|
||||
skip_special_tokens=True,
|
||||
).strip()
|
||||
print(answer)
|
||||
```
|
||||
|
||||
The reported benchmark result uses the five fixed official ITALIC
|
||||
demonstrations. Zero-shot use, different prompts, sampling, or another chat
|
||||
template should not be expected to reproduce the reported score.
|
||||
|
||||
## Model architecture
|
||||
|
||||
| Property | Value |
|
||||
|---|---|
|
||||
| Architecture | `LlamaForCausalLM` |
|
||||
| Model type | Llama-style decoder-only transformer |
|
||||
| Parameters | 437,760,960 |
|
||||
| Hidden size | 960 |
|
||||
| Layers | 32 |
|
||||
| Attention heads / KV heads | 15 / 5 |
|
||||
| Vocabulary size | 128,256 |
|
||||
| Maximum trained context | 2,048 tokens during SFT |
|
||||
| Checkpoint serialization | SafeTensors |
|
||||
|
||||
## Training
|
||||
|
||||
### Data
|
||||
|
||||
The final training stream was built online from pinned revisions of four public
|
||||
datasets:
|
||||
|
||||
| Dataset | Main contribution |
|
||||
|---|---|
|
||||
| `efederici/pinocchio` | Italian language, culture, and general-knowledge MCQs |
|
||||
| `FinancialSupport/italic_sft` | ITALIC-style Italian MCQs |
|
||||
| `FinancialSupport/italic_sft_ext` | Extended ITALIC-style MCQs |
|
||||
| `FinancialSupport/quiz_militare` | Italian civic and general-knowledge MCQs |
|
||||
|
||||
Rows were normalized, exactly deduplicated, split before upsampling, and
|
||||
decontaminated against the official ITALIC test and five demonstrations using
|
||||
character TF-IDF cosine and word-shingle MinHash/LSH. The final artifacts were:
|
||||
|
||||
| Artifact | Rows | Unique questions |
|
||||
|---|---:|---:|
|
||||
| Clean training stream | 108,288 | 67,654 |
|
||||
| Clean validation holdout | 6,745 | 6,745 |
|
||||
|
||||
The training stream contains 53,591 language draws and 54,697 culture draws.
|
||||
The benchmark labels were not used for checkpoint selection.
|
||||
|
||||
### Objective and hyperparameters
|
||||
|
||||
The model was trained for one epoch with completion-only supervised loss. Each
|
||||
target was rendered with five distinct training-pool demonstrations and
|
||||
independently permuted answer options. The official five ITALIC demonstrations
|
||||
were reserved for validation and final evaluation.
|
||||
|
||||
| Parameter | Value |
|
||||
|---|---:|
|
||||
| Hardware | 1x NVIDIA H100 |
|
||||
| Optimizer steps | 6,768 |
|
||||
| Batch size | 16 |
|
||||
| Initial learning rate | `3e-4` |
|
||||
| Schedule | 50-step warmup, cosine decay |
|
||||
| Weight decay | `0.0` |
|
||||
| Maximum sequence length | 2,048 |
|
||||
| Forward precision | BF16 autocast |
|
||||
| Parameters and optimizer updates | FP32 |
|
||||
| Gradient clipping | `1.0` |
|
||||
| Random seed | 0 |
|
||||
| Few-shot probability / count | `1.0 / 5` |
|
||||
| Option permutation probability | `1.0` |
|
||||
|
||||
The detached Modal pipeline took approximately 84 minutes end to end. Carbon
|
||||
emissions were not measured, so no emissions estimate is reported.
|
||||
|
||||
## Evaluation
|
||||
|
||||
The selected checkpoint was chosen at step 6,768 using only a clean validation
|
||||
holdout and the criterion
|
||||
`0.5 * language_accuracy + 0.5 * culture_accuracy`.
|
||||
|
||||
| Evaluation | Questions | Accuracy | Balanced accuracy | Unparsed |
|
||||
|---|---:|---:|---:|---:|
|
||||
| Validation holdout | 6,745 | 55.97% | 61.38% | 0 |
|
||||
| Official ITALIC fast five-shot, vLLM 0.26 | 10,000 | **44.35%** | not reported | 0 |
|
||||
|
||||
The official evaluation used greedy decoding, the official fast answer
|
||||
extractor, a pinned ITALIC harness, and the official five demonstrations. The
|
||||
result contained 4,435 correct answers out of 10,000.
|
||||
|
||||
## Reproducibility
|
||||
|
||||
The training and evaluation implementation is available in
|
||||
[`mattiacurri/zagreus-italic-challenge-xmoons`](https://github.com/mattiacurri/zagreus-italic-challenge-xmoons).
|
||||
The exact local source snapshot used for this release is identified by Git
|
||||
commit `afc6b7f09a1e107f38ae04358ece7ed85f6be7a3`.
|
||||
|
||||
Selected artifact identities:
|
||||
|
||||
```text
|
||||
d679f3e79221aea95c64ef0be177530b02dce25d4745b0247cbae7b677f1ceab training pool
|
||||
749dbcdea8b244c3609acecddf618d9780559f501da18b09f650dfe10f2fb881 validation pool
|
||||
0d534fd8eceec72b4fc3179d77afe7d6291766f5831bc4c8b69b9c45463c1656 selection.json
|
||||
2ac2f6e412ea1e8f67ce6d8395c1c9845270101590d84de1be748e5919833c93 official result JSON
|
||||
```
|
||||
|
||||
The source repository contains the full experiment and reproducibility reports.
|
||||
|
||||
## Limitations and risks
|
||||
|
||||
- The model is optimized for Italian multiple-choice prompts and often emits
|
||||
only an answer letter. It is not a broadly instruction-tuned chat model.
|
||||
- The 44.35% figure is one benchmark result under one exact prompt and backend
|
||||
configuration; it does not measure general Italian language competence.
|
||||
- Exact and fuzzy decontamination reduce known overlap but cannot prove the
|
||||
absence of semantic contamination or benchmark-distribution overfitting.
|
||||
- The public training sources may contain factual errors, social biases,
|
||||
stereotypes, or outdated information that can be inherited by the model.
|
||||
- Outputs outside the trained answer format may be unreliable.
|
||||
|
||||
## License and attribution
|
||||
|
||||
The base model weights are published under Apache-2.0. The included tokenizer
|
||||
and chat template come from a Llama 3 derivative and remain subject to the
|
||||
published Llama 3 terms; the metadata therefore uses the more restrictive
|
||||
`llama3` license identifier. Training datasets and benchmark assets retain
|
||||
their own terms. Some `FinancialSupport` dataset cards do not declare a license,
|
||||
so users should verify those terms before redistribution or commercial use.
|
||||
|
||||
## Acknowledgements
|
||||
|
||||
- [`firekern/zagreus-italic-challenge-xmoons`](https://github.com/firekern/zagreus-italic-challenge-xmoons), the upstream repository
|
||||
- [`mii-llm/zagreus-0.4B-ita`](https://huggingface.co/mii-llm/zagreus-0.4B-ita)
|
||||
- [`swap-uniba/LLaMAntino-3-ANITA-8B-Inst-DPO-ITA`](https://huggingface.co/swap-uniba/LLaMAntino-3-ANITA-8B-Inst-DPO-ITA)
|
||||
- [`mii-llm/Post-Training-Challenge`](https://huggingface.co/spaces/mii-llm/Post-Training-Challenge)
|
||||
- [ITALIC](https://italicbench.it/)
|
||||
5
chat_template.jinja
Normal file
5
chat_template.jinja
Normal file
@@ -0,0 +1,5 @@
|
||||
{% set loop_messages = messages %}{% for message in loop_messages %}{% set content = '<|start_header_id|>' + message['role'] + '<|end_header_id|>
|
||||
|
||||
'+ message['content'] | trim + '<|eot_id|>' %}{% if loop.index0 == 0 %}{% set content = bos_token + content %}{% endif %}{{ content }}{% endfor %}{% if add_generation_prompt %}{{ '<|start_header_id|>assistant<|end_header_id|>
|
||||
|
||||
' }}{% endif %}
|
||||
32
config.json
Normal file
32
config.json
Normal file
@@ -0,0 +1,32 @@
|
||||
{
|
||||
"architectures": [
|
||||
"LlamaForCausalLM"
|
||||
],
|
||||
"attention_bias": false,
|
||||
"attention_dropout": 0.0,
|
||||
"bos_token_id": 128000,
|
||||
"dtype": "float32",
|
||||
"eos_token_id": 128009,
|
||||
"head_dim": 64,
|
||||
"hidden_act": "silu",
|
||||
"hidden_size": 960,
|
||||
"initializer_range": 0.02,
|
||||
"intermediate_size": 2560,
|
||||
"max_position_embeddings": 4096,
|
||||
"mlp_bias": false,
|
||||
"model_type": "llama",
|
||||
"num_attention_heads": 15,
|
||||
"num_hidden_layers": 32,
|
||||
"num_key_value_heads": 5,
|
||||
"pad_token_id": null,
|
||||
"pretraining_tp": 1,
|
||||
"rms_norm_eps": 1e-05,
|
||||
"rope_parameters": {
|
||||
"rope_theta": 10000.0,
|
||||
"rope_type": "default"
|
||||
},
|
||||
"tie_word_embeddings": true,
|
||||
"transformers_version": "5.14.1",
|
||||
"use_cache": true,
|
||||
"vocab_size": 128256
|
||||
}
|
||||
6
generation_config.json
Normal file
6
generation_config.json
Normal file
@@ -0,0 +1,6 @@
|
||||
{
|
||||
"_from_model_config": true,
|
||||
"bos_token_id": 128000,
|
||||
"eos_token_id": 128009,
|
||||
"transformers_version": "5.14.1"
|
||||
}
|
||||
3
model.safetensors
Normal file
3
model.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:2d3d5c12f97ff9ffc4965c83442e38ad2e3c1c1b6b2216f37f217a6bbdc7a8bb
|
||||
size 1751076576
|
||||
129
selection.json
Normal file
129
selection.json
Normal file
@@ -0,0 +1,129 @@
|
||||
{
|
||||
"criterion": "0.5 * language_accuracy + 0.5 * culture_accuracy",
|
||||
"tie_breakers": [
|
||||
"fewer unparsed answers",
|
||||
"earlier checkpoint"
|
||||
],
|
||||
"training_pool": "data/prompts_200k_clean.jsonl",
|
||||
"training_pool_sha256": "d679f3e79221aea95c64ef0be177530b02dce25d4745b0247cbae7b677f1ceab",
|
||||
"validation_pool": "data/prompts_holdout_clean.jsonl",
|
||||
"validation_pool_sha256": "749dbcdea8b244c3609acecddf618d9780559f501da18b09f650dfe10f2fb881",
|
||||
"evaluation_steps": [
|
||||
1692,
|
||||
3384,
|
||||
5076,
|
||||
6768
|
||||
],
|
||||
"evaluations": [
|
||||
{
|
||||
"step": 1692,
|
||||
"rows": 6745,
|
||||
"correct": 3224,
|
||||
"accuracy": 0.47798369162342474,
|
||||
"balanced_accuracy": 0.5274351150178963,
|
||||
"macro": {
|
||||
"lang": {
|
||||
"correct": 716,
|
||||
"total": 1186,
|
||||
"accuracy": 0.6037099494097807
|
||||
},
|
||||
"culture": {
|
||||
"correct": 2508,
|
||||
"total": 5559,
|
||||
"accuracy": 0.4511602806260119
|
||||
}
|
||||
},
|
||||
"unparsed": 0,
|
||||
"prediction_distribution": {
|
||||
"A": 2534,
|
||||
"B": 1649,
|
||||
"C": 1840,
|
||||
"D": 630,
|
||||
"E": 92
|
||||
}
|
||||
},
|
||||
{
|
||||
"step": 3384,
|
||||
"rows": 6745,
|
||||
"correct": 3510,
|
||||
"accuracy": 0.5203854707190512,
|
||||
"balanced_accuracy": 0.5733892625695172,
|
||||
"macro": {
|
||||
"lang": {
|
||||
"correct": 777,
|
||||
"total": 1186,
|
||||
"accuracy": 0.6551433389544689
|
||||
},
|
||||
"culture": {
|
||||
"correct": 2733,
|
||||
"total": 5559,
|
||||
"accuracy": 0.4916351861845656
|
||||
}
|
||||
},
|
||||
"unparsed": 0,
|
||||
"prediction_distribution": {
|
||||
"A": 809,
|
||||
"B": 2210,
|
||||
"C": 2388,
|
||||
"D": 1310,
|
||||
"E": 28
|
||||
}
|
||||
},
|
||||
{
|
||||
"step": 5076,
|
||||
"rows": 6745,
|
||||
"correct": 3727,
|
||||
"accuracy": 0.5525574499629355,
|
||||
"balanced_accuracy": 0.6101524896048429,
|
||||
"macro": {
|
||||
"lang": {
|
||||
"correct": 829,
|
||||
"total": 1186,
|
||||
"accuracy": 0.6989881956155143
|
||||
},
|
||||
"culture": {
|
||||
"correct": 2898,
|
||||
"total": 5559,
|
||||
"accuracy": 0.5213167835941717
|
||||
}
|
||||
},
|
||||
"unparsed": 0,
|
||||
"prediction_distribution": {
|
||||
"A": 1529,
|
||||
"B": 1649,
|
||||
"C": 2189,
|
||||
"D": 1270,
|
||||
"E": 108
|
||||
}
|
||||
},
|
||||
{
|
||||
"step": 6768,
|
||||
"rows": 6745,
|
||||
"correct": 3775,
|
||||
"accuracy": 0.5596738324684952,
|
||||
"balanced_accuracy": 0.6138065310131664,
|
||||
"macro": {
|
||||
"lang": {
|
||||
"correct": 827,
|
||||
"total": 1186,
|
||||
"accuracy": 0.6973018549747049
|
||||
},
|
||||
"culture": {
|
||||
"correct": 2948,
|
||||
"total": 5559,
|
||||
"accuracy": 0.530311207051628
|
||||
}
|
||||
},
|
||||
"unparsed": 0,
|
||||
"prediction_distribution": {
|
||||
"A": 1625,
|
||||
"B": 1856,
|
||||
"C": 1861,
|
||||
"D": 1332,
|
||||
"E": 71
|
||||
}
|
||||
}
|
||||
],
|
||||
"selected_step": 6768,
|
||||
"selected_balanced_accuracy": 0.6138065310131664
|
||||
}
|
||||
3
tokenizer.json
Normal file
3
tokenizer.json
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:76ef6acdcaf0af01958639958ab6354f8578b8afcd7d76d6a124dd88514f9083
|
||||
size 17209016
|
||||
15
tokenizer_config.json
Normal file
15
tokenizer_config.json
Normal file
@@ -0,0 +1,15 @@
|
||||
{
|
||||
"backend": "tokenizers",
|
||||
"bos_token": "<|begin_of_text|>",
|
||||
"clean_up_tokenization_spaces": true,
|
||||
"eos_token": "<|eot_id|>",
|
||||
"is_local": true,
|
||||
"local_files_only": false,
|
||||
"model_input_names": [
|
||||
"input_ids",
|
||||
"attention_mask"
|
||||
],
|
||||
"model_max_length": 1000000000000000019884624838656,
|
||||
"pad_token": "<|eot_id|>",
|
||||
"tokenizer_class": "TokenizersBackend"
|
||||
}
|
||||
Reference in New Issue
Block a user