From bc49f1e12757cdb8e8c8d7646a12959f81149520 Mon Sep 17 00:00:00 2001 From: ModelHub XC Date: Sat, 12 Sep 2026 04:28:16 +0800 Subject: [PATCH] =?UTF-8?q?=E5=88=9D=E5=A7=8B=E5=8C=96=E9=A1=B9=E7=9B=AE?= =?UTF-8?q?=EF=BC=8C=E7=94=B1ModelHub=20XC=E7=A4=BE=E5=8C=BA=E6=8F=90?= =?UTF-8?q?=E4=BE=9B=E6=A8=A1=E5=9E=8B?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Model: fgoose180/zagreus-0.4B-xmoons-improved Source: Original Platform --- .gitattributes | 37 +++++++ README.md | 242 +++++++++++++++++++++++++++++++++++++++++ chat_template.jinja | 5 + config.json | 32 ++++++ generation_config.json | 6 + model.safetensors | 3 + selection.json | 129 ++++++++++++++++++++++ tokenizer.json | 3 + tokenizer_config.json | 15 +++ 9 files changed, 472 insertions(+) create mode 100644 .gitattributes create mode 100644 README.md create mode 100644 chat_template.jinja create mode 100644 config.json create mode 100644 generation_config.json create mode 100644 model.safetensors create mode 100644 selection.json create mode 100644 tokenizer.json create mode 100644 tokenizer_config.json diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..64bc6a8 --- /dev/null +++ b/.gitattributes @@ -0,0 +1,37 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text +tokenizer.json filter=lfs diff=lfs merge=lfs -text +zagreus-final/tokenizer.json filter=lfs diff=lfs merge=lfs -text diff --git a/README.md b/README.md new file mode 100644 index 0000000..80bfeb0 --- /dev/null +++ b/README.md @@ -0,0 +1,242 @@ +--- +library_name: transformers +pipeline_tag: text-generation +language: + - it +license: llama3 +base_model: + - mii-llm/zagreus-0.4B-ita +datasets: + - efederici/pinocchio + - FinancialSupport/italic_sft + - FinancialSupport/italic_sft_ext + - FinancialSupport/quiz_militare +tags: + - llama + - italian + - small-language-model + - multiple-choice + - supervised-fine-tuning + - post-training + - italic +--- + +# Zagreus 0.4B xmoons improved + +Zagreus 0.4B xmoons improved is an Italian multiple-choice model fine-tuned +from [`mii-llm/zagreus-0.4B-ita`](https://huggingface.co/mii-llm/zagreus-0.4B-ita). +It is the selected checkpoint from the reproducible H100 run that reached +**44.35% accuracy** on the official 10,000-question ITALIC benchmark in fast, +five-shot mode through vLLM. + +The model was developed for the +[`mii-llm/Post-Training-Challenge`](https://huggingface.co/spaces/mii-llm/Post-Training-Challenge). + +The model is a compact Llama-style causal language model with 437,760,960 +parameters. Its tokenizer and Llama 3 chat template come from +[`swap-uniba/LLaMAntino-3-ANITA-8B-Inst-DPO-ITA`](https://huggingface.co/swap-uniba/LLaMAntino-3-ANITA-8B-Inst-DPO-ITA). +The repository is a self-contained Transformers checkpoint and uses +`<|eot_id|>` as both the end-of-turn and generation stop token. + +## Intended use + +The model is intended for: + +- research on compact Italian language models; +- Italian multiple-choice question answering; +- participation in the MII Post-Training Challenge; +- reproduction and analysis of the ITALIC post-training experiment. + +It is not designed as a general-purpose assistant, a factual authority, or a +component for medical, legal, financial, or other high-stakes decisions. + +## Usage + +Install PyTorch and Transformers, then load the checkpoint from the repository +root: + +```python +import torch +from transformers import AutoModelForCausalLM, AutoTokenizer + +model_id = "fgoose180/zagreus-0.4B-xmoons-improved" +device = "cuda" if torch.cuda.is_available() else "cpu" +dtype = torch.bfloat16 if device == "cuda" else torch.float32 + +tokenizer = AutoTokenizer.from_pretrained(model_id) +model = AutoModelForCausalLM.from_pretrained( + model_id, + dtype=dtype, +).to(device) + +messages = [ + {"role": "system", "content": "Sei un assistente utile."}, + { + "role": "user", + "content": """Rispondi alla seguente domanda a scelta multipla sull'argomento 'data_geography'. La tua risposta deve essere nel seguente formato: 'LETTERA' (senza virgolette) dove LETTERA è una tra ABCD. Scrivi solo la lettera corrispondente alla tua risposta senza spiegazioni. + +Qual è la capitale d'Italia? + +A) Roma +B) Milano +C) Torino +D) Napoli + +Risposta:""", + }, +] + +inputs = tokenizer.apply_chat_template( + messages, + add_generation_prompt=True, + return_tensors="pt", + return_dict=True, +).to(device) + +with torch.inference_mode(): + output = model.generate( + **inputs, + max_new_tokens=8, + do_sample=False, + eos_token_id=tokenizer.eos_token_id, + pad_token_id=tokenizer.pad_token_id, + ) + +answer = tokenizer.decode( + output[0, inputs["input_ids"].shape[1] :], + skip_special_tokens=True, +).strip() +print(answer) +``` + +The reported benchmark result uses the five fixed official ITALIC +demonstrations. Zero-shot use, different prompts, sampling, or another chat +template should not be expected to reproduce the reported score. + +## Model architecture + +| Property | Value | +|---|---| +| Architecture | `LlamaForCausalLM` | +| Model type | Llama-style decoder-only transformer | +| Parameters | 437,760,960 | +| Hidden size | 960 | +| Layers | 32 | +| Attention heads / KV heads | 15 / 5 | +| Vocabulary size | 128,256 | +| Maximum trained context | 2,048 tokens during SFT | +| Checkpoint serialization | SafeTensors | + +## Training + +### Data + +The final training stream was built online from pinned revisions of four public +datasets: + +| Dataset | Main contribution | +|---|---| +| `efederici/pinocchio` | Italian language, culture, and general-knowledge MCQs | +| `FinancialSupport/italic_sft` | ITALIC-style Italian MCQs | +| `FinancialSupport/italic_sft_ext` | Extended ITALIC-style MCQs | +| `FinancialSupport/quiz_militare` | Italian civic and general-knowledge MCQs | + +Rows were normalized, exactly deduplicated, split before upsampling, and +decontaminated against the official ITALIC test and five demonstrations using +character TF-IDF cosine and word-shingle MinHash/LSH. The final artifacts were: + +| Artifact | Rows | Unique questions | +|---|---:|---:| +| Clean training stream | 108,288 | 67,654 | +| Clean validation holdout | 6,745 | 6,745 | + +The training stream contains 53,591 language draws and 54,697 culture draws. +The benchmark labels were not used for checkpoint selection. + +### Objective and hyperparameters + +The model was trained for one epoch with completion-only supervised loss. Each +target was rendered with five distinct training-pool demonstrations and +independently permuted answer options. The official five ITALIC demonstrations +were reserved for validation and final evaluation. + +| Parameter | Value | +|---|---:| +| Hardware | 1x NVIDIA H100 | +| Optimizer steps | 6,768 | +| Batch size | 16 | +| Initial learning rate | `3e-4` | +| Schedule | 50-step warmup, cosine decay | +| Weight decay | `0.0` | +| Maximum sequence length | 2,048 | +| Forward precision | BF16 autocast | +| Parameters and optimizer updates | FP32 | +| Gradient clipping | `1.0` | +| Random seed | 0 | +| Few-shot probability / count | `1.0 / 5` | +| Option permutation probability | `1.0` | + +The detached Modal pipeline took approximately 84 minutes end to end. Carbon +emissions were not measured, so no emissions estimate is reported. + +## Evaluation + +The selected checkpoint was chosen at step 6,768 using only a clean validation +holdout and the criterion +`0.5 * language_accuracy + 0.5 * culture_accuracy`. + +| Evaluation | Questions | Accuracy | Balanced accuracy | Unparsed | +|---|---:|---:|---:|---:| +| Validation holdout | 6,745 | 55.97% | 61.38% | 0 | +| Official ITALIC fast five-shot, vLLM 0.26 | 10,000 | **44.35%** | not reported | 0 | + +The official evaluation used greedy decoding, the official fast answer +extractor, a pinned ITALIC harness, and the official five demonstrations. The +result contained 4,435 correct answers out of 10,000. + +## Reproducibility + +The training and evaluation implementation is available in +[`mattiacurri/zagreus-italic-challenge-xmoons`](https://github.com/mattiacurri/zagreus-italic-challenge-xmoons). +The exact local source snapshot used for this release is identified by Git +commit `afc6b7f09a1e107f38ae04358ece7ed85f6be7a3`. + +Selected artifact identities: + +```text +d679f3e79221aea95c64ef0be177530b02dce25d4745b0247cbae7b677f1ceab training pool +749dbcdea8b244c3609acecddf618d9780559f501da18b09f650dfe10f2fb881 validation pool +0d534fd8eceec72b4fc3179d77afe7d6291766f5831bc4c8b69b9c45463c1656 selection.json +2ac2f6e412ea1e8f67ce6d8395c1c9845270101590d84de1be748e5919833c93 official result JSON +``` + +The source repository contains the full experiment and reproducibility reports. + +## Limitations and risks + +- The model is optimized for Italian multiple-choice prompts and often emits + only an answer letter. It is not a broadly instruction-tuned chat model. +- The 44.35% figure is one benchmark result under one exact prompt and backend + configuration; it does not measure general Italian language competence. +- Exact and fuzzy decontamination reduce known overlap but cannot prove the + absence of semantic contamination or benchmark-distribution overfitting. +- The public training sources may contain factual errors, social biases, + stereotypes, or outdated information that can be inherited by the model. +- Outputs outside the trained answer format may be unreliable. + +## License and attribution + +The base model weights are published under Apache-2.0. The included tokenizer +and chat template come from a Llama 3 derivative and remain subject to the +published Llama 3 terms; the metadata therefore uses the more restrictive +`llama3` license identifier. Training datasets and benchmark assets retain +their own terms. Some `FinancialSupport` dataset cards do not declare a license, +so users should verify those terms before redistribution or commercial use. + +## Acknowledgements + +- [`firekern/zagreus-italic-challenge-xmoons`](https://github.com/firekern/zagreus-italic-challenge-xmoons), the upstream repository +- [`mii-llm/zagreus-0.4B-ita`](https://huggingface.co/mii-llm/zagreus-0.4B-ita) +- [`swap-uniba/LLaMAntino-3-ANITA-8B-Inst-DPO-ITA`](https://huggingface.co/swap-uniba/LLaMAntino-3-ANITA-8B-Inst-DPO-ITA) +- [`mii-llm/Post-Training-Challenge`](https://huggingface.co/spaces/mii-llm/Post-Training-Challenge) +- [ITALIC](https://italicbench.it/) diff --git a/chat_template.jinja b/chat_template.jinja new file mode 100644 index 0000000..39bd0c9 --- /dev/null +++ b/chat_template.jinja @@ -0,0 +1,5 @@ +{% set loop_messages = messages %}{% for message in loop_messages %}{% set content = '<|start_header_id|>' + message['role'] + '<|end_header_id|> + +'+ message['content'] | trim + '<|eot_id|>' %}{% if loop.index0 == 0 %}{% set content = bos_token + content %}{% endif %}{{ content }}{% endfor %}{% if add_generation_prompt %}{{ '<|start_header_id|>assistant<|end_header_id|> + +' }}{% endif %} \ No newline at end of file diff --git a/config.json b/config.json new file mode 100644 index 0000000..b34db0d --- /dev/null +++ b/config.json @@ -0,0 +1,32 @@ +{ + "architectures": [ + "LlamaForCausalLM" + ], + "attention_bias": false, + "attention_dropout": 0.0, + "bos_token_id": 128000, + "dtype": "float32", + "eos_token_id": 128009, + "head_dim": 64, + "hidden_act": "silu", + "hidden_size": 960, + "initializer_range": 0.02, + "intermediate_size": 2560, + "max_position_embeddings": 4096, + "mlp_bias": false, + "model_type": "llama", + "num_attention_heads": 15, + "num_hidden_layers": 32, + "num_key_value_heads": 5, + "pad_token_id": null, + "pretraining_tp": 1, + "rms_norm_eps": 1e-05, + "rope_parameters": { + "rope_theta": 10000.0, + "rope_type": "default" + }, + "tie_word_embeddings": true, + "transformers_version": "5.14.1", + "use_cache": true, + "vocab_size": 128256 +} diff --git a/generation_config.json b/generation_config.json new file mode 100644 index 0000000..f2ce98b --- /dev/null +++ b/generation_config.json @@ -0,0 +1,6 @@ +{ + "_from_model_config": true, + "bos_token_id": 128000, + "eos_token_id": 128009, + "transformers_version": "5.14.1" +} diff --git a/model.safetensors b/model.safetensors new file mode 100644 index 0000000..da34e06 --- /dev/null +++ b/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2d3d5c12f97ff9ffc4965c83442e38ad2e3c1c1b6b2216f37f217a6bbdc7a8bb +size 1751076576 diff --git a/selection.json b/selection.json new file mode 100644 index 0000000..ed33b1e --- /dev/null +++ b/selection.json @@ -0,0 +1,129 @@ +{ + "criterion": "0.5 * language_accuracy + 0.5 * culture_accuracy", + "tie_breakers": [ + "fewer unparsed answers", + "earlier checkpoint" + ], + "training_pool": "data/prompts_200k_clean.jsonl", + "training_pool_sha256": "d679f3e79221aea95c64ef0be177530b02dce25d4745b0247cbae7b677f1ceab", + "validation_pool": "data/prompts_holdout_clean.jsonl", + "validation_pool_sha256": "749dbcdea8b244c3609acecddf618d9780559f501da18b09f650dfe10f2fb881", + "evaluation_steps": [ + 1692, + 3384, + 5076, + 6768 + ], + "evaluations": [ + { + "step": 1692, + "rows": 6745, + "correct": 3224, + "accuracy": 0.47798369162342474, + "balanced_accuracy": 0.5274351150178963, + "macro": { + "lang": { + "correct": 716, + "total": 1186, + "accuracy": 0.6037099494097807 + }, + "culture": { + "correct": 2508, + "total": 5559, + "accuracy": 0.4511602806260119 + } + }, + "unparsed": 0, + "prediction_distribution": { + "A": 2534, + "B": 1649, + "C": 1840, + "D": 630, + "E": 92 + } + }, + { + "step": 3384, + "rows": 6745, + "correct": 3510, + "accuracy": 0.5203854707190512, + "balanced_accuracy": 0.5733892625695172, + "macro": { + "lang": { + "correct": 777, + "total": 1186, + "accuracy": 0.6551433389544689 + }, + "culture": { + "correct": 2733, + "total": 5559, + "accuracy": 0.4916351861845656 + } + }, + "unparsed": 0, + "prediction_distribution": { + "A": 809, + "B": 2210, + "C": 2388, + "D": 1310, + "E": 28 + } + }, + { + "step": 5076, + "rows": 6745, + "correct": 3727, + "accuracy": 0.5525574499629355, + "balanced_accuracy": 0.6101524896048429, + "macro": { + "lang": { + "correct": 829, + "total": 1186, + "accuracy": 0.6989881956155143 + }, + "culture": { + "correct": 2898, + "total": 5559, + "accuracy": 0.5213167835941717 + } + }, + "unparsed": 0, + "prediction_distribution": { + "A": 1529, + "B": 1649, + "C": 2189, + "D": 1270, + "E": 108 + } + }, + { + "step": 6768, + "rows": 6745, + "correct": 3775, + "accuracy": 0.5596738324684952, + "balanced_accuracy": 0.6138065310131664, + "macro": { + "lang": { + "correct": 827, + "total": 1186, + "accuracy": 0.6973018549747049 + }, + "culture": { + "correct": 2948, + "total": 5559, + "accuracy": 0.530311207051628 + } + }, + "unparsed": 0, + "prediction_distribution": { + "A": 1625, + "B": 1856, + "C": 1861, + "D": 1332, + "E": 71 + } + } + ], + "selected_step": 6768, + "selected_balanced_accuracy": 0.6138065310131664 +} diff --git a/tokenizer.json b/tokenizer.json new file mode 100644 index 0000000..3350de7 --- /dev/null +++ b/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:76ef6acdcaf0af01958639958ab6354f8578b8afcd7d76d6a124dd88514f9083 +size 17209016 diff --git a/tokenizer_config.json b/tokenizer_config.json new file mode 100644 index 0000000..b76464f --- /dev/null +++ b/tokenizer_config.json @@ -0,0 +1,15 @@ +{ + "backend": "tokenizers", + "bos_token": "<|begin_of_text|>", + "clean_up_tokenization_spaces": true, + "eos_token": "<|eot_id|>", + "is_local": true, + "local_files_only": false, + "model_input_names": [ + "input_ids", + "attention_mask" + ], + "model_max_length": 1000000000000000019884624838656, + "pad_token": "<|eot_id|>", + "tokenizer_class": "TokenizersBackend" +}