commit 1b8467a18ef4bae9b668cc6c339da13d1d93e1b6 Author: ModelHub XC Date: Wed Sep 9 00:04:17 2026 +0800 初始化项目,由ModelHub XC社区提供模型 Model: giux78/zagreus_0.4_competition Source: Original Platform diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..52373fe --- /dev/null +++ b/.gitattributes @@ -0,0 +1,36 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text +tokenizer.json filter=lfs diff=lfs merge=lfs -text diff --git a/README.md b/README.md new file mode 100644 index 0000000..8aee86d --- /dev/null +++ b/README.md @@ -0,0 +1,58 @@ +--- +base_model: mii-llm/nesso-0.4B-agentic +language: + - it +pipeline_tag: text-generation +tags: + - distillation + - on-policy-distillation + - italic + - italian + - mcqa +--- + +# zagreus_0.4_competition + +**nesso-0.4B-agentic improved on the [ITALIC](https://github.com/Crisp-Unimib/ITALIC) benchmark by on-policy distillation from Coloss/nesso-3B.** + +Official ITALIC harness (`run_eval.py`, full 10,000 questions, 5-shot): + +| Model | fast | slow (CoT) | +|---|---:|---:| +| mii-llm/nesso-0.4B-agentic (baseline) | 33.1% | 0.0%* | +| **this model** | **37.2%** | **33.2%** | +| Coloss/nesso-3B (teacher) | 50.7% | 2.5%* | + +\* baseline/teacher answer with a bare letter, which the official slow-mode extractor cannot parse; this model answers parseably in both modes. + +## Method + +On-policy distillation ([tinker-cookbook recipe](https://github.com/thinking-machines-lab/tinker-cookbook/blob/main/tinker_cookbook/recipes/distillation/on_policy_distillation.py), reimplemented in [mii-llm/palingenesis](https://github.com/mii-llm/palingenesis) branch `odp`): every step the student samples completions with its current weights and the loss is the full-distribution reverse KL to the teacher over exactly those tokens. Student (ChatML) and teacher (Llama-3 template) share the Llama-3 base vocabulary; prompts are rendered per-model and the student's `<|im_end|>` mass is merged into the teacher's `<|eot_id|>` slot, so stopping behavior is distilled too. + +The winning run (`opd_v3`, checkpoint step 550 of 600): + +- pool of 49,643 ITALIC-format MCQA prompts (pinocchio-raw + MMLU-ita sources, deduped against the ITALIC 10k test set by normalized-question hash), **filtered to rows the teacher answers correctly** (`pgs distill-score` — pure KL would distill the teacher's errors too) +- Italian-language (vocabulary/grammar/comprehension) rows upweighted ×4 +- 80% of prompts rendered with ITALIC's official 5 shots, `max_new_tokens 8` (terse supervision) +- 600 steps, batch 32 prompts, lr 1e-5 cosine, full reverse KL, ~40 min on one A100 80GB + +Training/eval code: [giux78/zagreus_0.4_competition](https://github.com/giux78/zagreus_0.4_competition) (experiment) + [mii-llm/palingenesis](https://github.com/mii-llm/palingenesis) `odp` branch (library, `pgs distill`). + +## Usage + +```python +from transformers import AutoModelForCausalLM, AutoTokenizer + +tok = AutoTokenizer.from_pretrained("giux78/zagreus_0.4_competition") +model = AutoModelForCausalLM.from_pretrained("giux78/zagreus_0.4_competition", dtype="bfloat16") + +messages = [ + {"role": "system", "content": "Sei un assistente utile."}, + {"role": "user", "content": "Rispondi alla seguente domanda a scelta multipla..."}, +] +ids = tok.apply_chat_template(messages, add_generation_prompt=True, return_tensors="pt") +out = model.generate(ids, max_new_tokens=8, do_sample=False) +print(tok.decode(out[0, ids.shape[1]:], skip_special_tokens=True)) +``` + +For ITALIC-style evaluation use `max_new_tokens=8` and greedy decoding (the model answers with the bare option letter). diff --git a/chat_template.jinja b/chat_template.jinja new file mode 100644 index 0000000..e129b5d --- /dev/null +++ b/chat_template.jinja @@ -0,0 +1,68 @@ +{{ bos_token }} +{% if tools %} +<|im_start|>system +{% if messages and messages[0]['role'] == 'system' %} +{{ messages[0]['content'] }} +{% endif %} +# Tools +You may call one or more functions to assist with the user query. +You are provided with function signatures within XML tags: + +{% for tool in tools %} +{{ tool | tojson }} +{% endfor %} + +For each function call, return a json object with function name and arguments within XML tags: + +{"name": , "arguments": } +<|im_end|> +{% else %} +{% if messages and messages[0]['role'] == 'system' %} +<|im_start|>system +{{ messages[0]['content'] }}<|im_end|> +{% endif %} +{% endif %} + +{% for message in messages %} +{% if message.content is string %} + {% set content = message.content %} +{% else %} + {% set content = '' %} +{% endif %} + +{% if (message.role == 'user') or (message.role == 'system' and not loop.first) %} +<|im_start|>{{ message.role }} +{{ content }}<|im_end|> + +{% elif message.role == 'assistant' %} +<|im_start|>assistant +{{ content }} +{% if message.tool_calls %} + {% for tool_call in message.tool_calls %} + {% if tool_call.function %} + {% set tool_call = tool_call.function %} + {% endif %} + +{"name": "{{ tool_call.name }}", "arguments": {% if tool_call.arguments is string %}{{ tool_call.arguments }}{% else %}{{ tool_call.arguments | tojson }}{% endif %}} + + {% endfor %} +{% endif %} +<|im_end|> + +{% elif message.role == 'tool' %} +{% if loop.first or (messages[loop.index0 - 1].role != 'tool') %} +<|im_start|>user +{% endif %} + +{{ content }} + +{% if loop.last or (messages[loop.index0 + 1].role != 'tool') %} +<|im_end|> +{% endif %} + +{% endif %} +{% endfor %} + +{% if add_generation_prompt %} +<|im_start|>assistant +{% endif %} diff --git a/config.json b/config.json new file mode 100644 index 0000000..29778e4 --- /dev/null +++ b/config.json @@ -0,0 +1,32 @@ +{ + "architectures": [ + "LlamaForCausalLM" + ], + "attention_bias": false, + "attention_dropout": 0.0, + "bos_token_id": 128000, + "dtype": "float32", + "eos_token_id": 128256, + "head_dim": 64, + "hidden_act": "silu", + "hidden_size": 960, + "initializer_range": 0.02, + "intermediate_size": 2560, + "max_position_embeddings": 4096, + "mlp_bias": false, + "model_type": "llama", + "num_attention_heads": 15, + "num_hidden_layers": 32, + "num_key_value_heads": 5, + "pad_token_id": null, + "pretraining_tp": 1, + "rms_norm_eps": 1e-05, + "rope_parameters": { + "rope_theta": 10000.0, + "rope_type": "default" + }, + "tie_word_embeddings": true, + "transformers_version": "5.7.0", + "use_cache": false, + "vocab_size": 128262 +} diff --git a/generation_config.json b/generation_config.json new file mode 100644 index 0000000..e9e395f --- /dev/null +++ b/generation_config.json @@ -0,0 +1,7 @@ +{ + "_from_model_config": true, + "bos_token_id": 128000, + "do_sample": true, + "eos_token_id": 128001, + "transformers_version": "5.7.0" +} diff --git a/model.safetensors b/model.safetensors new file mode 100644 index 0000000..421086c --- /dev/null +++ b/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f744242923594afe1764865ef90a2ef03c76d32b18c1046a094798efc3b89998 +size 1751099616 diff --git a/opd_config.json b/opd_config.json new file mode 100644 index 0000000..7ce056c --- /dev/null +++ b/opd_config.json @@ -0,0 +1,56 @@ +{ + "model": { + "student": "mii-llm/nesso-0.4B-agentic", + "teacher": "giux78/nesso-3B", + "teacher_device": "", + "gradient_checkpointing": true + }, + "bridge": { + "eos_map": { + "<|im_end|>": "<|eot_id|>" + }, + "extra_stop_tokens": [ + "<|end_of_text|>" + ], + "probe_texts": [] + }, + "data": { + "prompts_path": "/home/ecuser/ai/zagreus_competition/data/prompts_v3.jsonl", + "shots_path": "/home/ecuser/ai/zagreus_competition/data/5_shots.jsonl", + "dev_size": 500, + "p_reference_shots": 0.8, + "p_pool_shots": 0.1, + "pool_shots_max_k": 5, + "system_message": "" + }, + "sampling": { + "batch_prompts": 32, + "group_size": 1, + "temperature": 1.0, + "max_new_tokens": 8, + "cot_fraction": 0.0, + "cot_max_new_tokens": 300, + "gen_micro_seqs": 64 + }, + "train": { + "output_dir": "/home/ecuser/ai/palingenesis/runs/opd_v3", + "steps": 600, + "learning_rate": 1e-05, + "warmup_steps": 50, + "lr_scheduler": "cosine", + "max_grad_norm": 1.0, + "loss_fn": "full_kl", + "seed": 0, + "score_micro_seqs": 8, + "eval_every": 50, + "eval_dev_samples": 200, + "save_steps": 50, + "keep_checkpoints": 0 + }, + "logging": { + "log_every": 10, + "use_wandb": false, + "project": "palingenesis-opd", + "run_name": "" + } +} \ No newline at end of file diff --git a/tokenizer.json b/tokenizer.json new file mode 100644 index 0000000..980d1f7 --- /dev/null +++ b/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4a80502037c38583156839ad1486db1e7e5aa0b3f9d47e072f16e6b30a0eb2dd +size 17211058 diff --git a/tokenizer_config.json b/tokenizer_config.json new file mode 100644 index 0000000..31238ff --- /dev/null +++ b/tokenizer_config.json @@ -0,0 +1,15 @@ +{ + "backend": "tokenizers", + "bos_token": "<|begin_of_text|>", + "clean_up_tokenization_spaces": true, + "eos_token": "<|im_end|>", + "is_local": false, + "local_files_only": false, + "model_input_names": [ + "input_ids", + "attention_mask" + ], + "model_max_length": 131072, + "pad_token": "<|end_of_text|>", + "tokenizer_class": "TokenizersBackend" +}