commit 6561c9acbb612096ca6692361655af8d2abb1b61 Author: ModelHub XC Date: Tue Sep 29 20:46:26 2026 +0800 初始化项目,由ModelHub XC社区提供模型 Model: minnesotanlp/Finch-8B-KTO Source: Original Platform diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..017382c --- /dev/null +++ b/.gitattributes @@ -0,0 +1,38 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text +tokenizer.json filter=lfs diff=lfs merge=lfs -text +assets/intro_teaser.png filter=lfs diff=lfs merge=lfs -text +assets/results_kto.png filter=lfs diff=lfs merge=lfs -text diff --git a/README.md b/README.md new file mode 100644 index 0000000..7aaa7f2 --- /dev/null +++ b/README.md @@ -0,0 +1,210 @@ +--- +license: apache-2.0 +language: +- en +library_name: transformers +pipeline_tag: text-generation +base_model: +- minnesotanlp/Finch-8B +datasets: +- minnesotanlp/Finch-Collection +tags: +- evolution-fine-tuning +- evolutionary-search +- discovery +- preference-learning +- kto +- mutation-operator +- mid-training +--- + +
+ + +

+ Evolution Fine-Tuning: Learning to Discover Across 371 Optimization Tasks +

+

+ A mid-training "practice phase" that teaches small open-source LLMs how to evolve solutions. +

+ +

+ Website + + arXiv + + GitHub + Dataset + Finch-2B + Finch-4B + Finch-8B + Finch-9B + Finch-4B-KTO + Apache 2.0 +

+ +
+ +**Finch-8B-KTO** extends [**`Finch-8B`**](https://huggingface.co/minnesotanlp/Finch-8B) with a second, preference-learning stage ([**KTO**](https://arxiv.org/abs/2402.01306)). On top of evolution fine-tuning — which teaches the model *how to evolve a solution* as a **mutation operator** — KTO adds the ability to **self-judge** which candidate solutions are promising and which fall short. It is the family's **strongest offline-RL variant**, surpassing the best human score on multiple mathematical-discovery tasks. Lineage: **Qwen3-8B → Finch-8B (EFT + SFT) → Finch-8B-KTO**. + +## TL;DR + +Evolution Fine-Tuning (EFT) turns evolutionary search **trajectories** into supervision, moving discovery behavior from the *scaffold* into the *model*. **KTO** then trains the model on `improved` vs. `regressed` transitions jointly, so it internalizes a sense of solution quality — pushing Finch-8B past the best human score on both autocorrelation-inequality tasks. + +
+ EFT as mid-training +
+ +- (Left) EFT acts as mid-training, boosting Finch's discovery on the Erdős minimum-overlap problem under both test-time search and test-time learning. +- (Right) On NP-hard competitive programming, Finch composes strategies learned across diverse domains, while the base model relies on a single repetitive strategy. + +## Finch family + +| Model | Base | Params | Training | 🤗 Hugging Face | +|---|---|---:|---|:---:| +| `Finch-2B` | Qwen3.5-2B | 2B | EFT | [![Open on Hugging Face](https://img.shields.io/badge/-Open-FFD21E?logo=huggingface&logoColor=black)](https://huggingface.co/minnesotanlp/Finch-2B) | +| `Finch-4B` | Qwen3.5-4B | 4B | EFT | [![Open on Hugging Face](https://img.shields.io/badge/-Open-FFD21E?logo=huggingface&logoColor=black)](https://huggingface.co/minnesotanlp/Finch-4B) | +| `Finch-8B` | Qwen3-8B | 8B | EFT | [![Open on Hugging Face](https://img.shields.io/badge/-Open-FFD21E?logo=huggingface&logoColor=black)](https://huggingface.co/minnesotanlp/Finch-8B) | +| `Finch-9B` | Qwen3.5-9B | 9B | EFT | [![Open on Hugging Face](https://img.shields.io/badge/-Open-FFD21E?logo=huggingface&logoColor=black)](https://huggingface.co/minnesotanlp/Finch-9B) | +| `Finch-4B-KTO` | Qwen3.5-4B | 4B | EFT + KTO | [![Open on Hugging Face](https://img.shields.io/badge/-Open-FFD21E?logo=huggingface&logoColor=black)](https://huggingface.co/minnesotanlp/Finch-4B-KTO) | +| **`Finch-8B-KTO`** ← *this model* | **Qwen3-8B** | **8B** | **EFT + KTO** | [![Open on Hugging Face](https://img.shields.io/badge/-Open-FFD21E?logo=huggingface&logoColor=black)](https://huggingface.co/minnesotanlp/Finch-8B-KTO) | + +## How to Use Finch + +1. **Execute OpenEvolve scaffold with Finch (vLLM serving, recommended)** + +Finch is a **mutation operator for evolutionary search**, most effective driven by a scaffold such as **OpenEvolve** (`T = 100`, temperature `0.7`, top-`p` `0.95`, up to `30K` tokens). +You can also use other scaffolds in the [SkyDiscover](https://github.com/skydiscover-ai/skydiscover) framework, but we do not guarantee performance, as our model is trained on OpenEvolve's trajectories — one of this work's limitations. + +2. **Calling Finch directly** + +You can also call Finch directly: + +**System prompt** (task-level instruction from the OpenEvolve scaffold): +``` +You are an expert mathematician specializing in circle packing problems and computational geometry. +Your task is to improve a constructor function that directly produces a specific arrangement of +26 circles in a unit square, maximizing the sum of their radii. +The AlphaEvolve paper achieved a sum of 2.635 for n=26. + +Key geometric insights: +- Circle packings often follow hexagonal patterns in the densest regions +- Maximum density for infinite circle packing is pi/(2*sqrt(3)) ≈ 0.9069 +- Edge effects make square container packing harder than infinite packing +- Similar radius circles often form regular patterns, while varied radii allow better space utilization +``` + +**User prompt** (evolutionary state — current program + evaluator feedback + evolutionary history): +``` +# Current Program Information +- Fitness: 0.3642 (sum_radii: 0.9598) +- Focus areas: Fitness unchanged at 0.3642. Consider simplifying — code length exceeds 500 characters. + +# Program Evolution History +## Previous Attempts + +### Attempt 1 +- Changes: Replace concentric ring placement with hexagonal lattice (5-6-5-6-5 row pattern) +- Metrics: sum_radii: 0.9598, validity: 1.0 — Improvement in all metrics + +# Current Program + +# EVOLVE-BLOCK-START +import numpy as np + +def construct_packing(): + n = 26 + centers = np.zeros((n, 2)) + centers[0] = [0.5, 0.5] # center circle + for i in range(8): # inner ring + angle = 2 * np.pi * i / 8 + centers[i+1] = [0.5 + 0.3*np.cos(angle), 0.5 + 0.3*np.sin(angle)] + for i in range(16): # outer ring + angle = 2 * np.pi * i / 16 + centers[i+9] = [0.5 + 0.7*np.cos(angle), 0.5 + 0.7*np.sin(angle)] + centers = np.clip(centers, 0.01, 0.99) + radii = compute_max_radii(centers) + return centers, radii, np.sum(radii) +# EVOLVE-BLOCK-END +``` + +```python +import torch +from transformers import AutoModelForCausalLM, AutoTokenizer + +model_id = "minnesotanlp/Finch-8B-KTO" +tokenizer = AutoTokenizer.from_pretrained(model_id) +model = AutoModelForCausalLM.from_pretrained(model_id, torch_dtype="auto", device_map="auto") + +# Given an evolutionary state — task instruction + parent program + evolutionary history +# + evaluator feedback — Finch proposes an improved candidate program. +messages = [ + {"role": "system", "content": SYSTEM_PROMPT}, # provided by your evolutionary scaffold + {"role": "user", "content": USER_PROMPT}, # parent program + feedback + history +] +inputs = tokenizer.apply_chat_template( + messages, add_generation_prompt=True, return_tensors="pt" +).to(model.device) + +out = model.generate(inputs, max_new_tokens=30000, do_sample=True, temperature=0.7, top_p=0.95) +print(tokenizer.decode(out[0][inputs.shape[-1]:], skip_special_tokens=True)) +``` + +## Training + +Two stages: + +- **Stage 1 — EFT (SFT).** [`Finch-8B`](https://huggingface.co/minnesotanlp/Finch-8B): full SFT of **Qwen3-8B** on `improved` transitions from the [Finch Collection](https://huggingface.co/datasets/minnesotanlp/Finch-Collection) (355 training tasks; one run/task → **30,445** examples; 900 for validation) with [LLaMA-Factory](https://github.com/hiyouga/LLaMA-Factory) — 1 epoch, global batch size 128, LR 1e-5, on 8× NVIDIA H200 140GB. +- **Stage 2 — KTO.** Preference learning ([KTO](https://arxiv.org/abs/2402.01306)) on `improved` (desirable) and `regressed` (undesirable) transitions jointly, maximizing the contrastive signal that guides the model toward self-judging which solutions are promising and which fall short. +- **Teacher (data).** Trajectories generated by **Qwen3.5-397B-A17B** inside the **OpenEvolve** scaffold. + +## Results + +- Finch outperforms its same-size base model by +10.2% on 22 held-out tasks across 5 domains, with improvements of up to +290% on individual tasks. +- Larger models benefit more, and Finch-4B matches a model roughly 2× larger on the Erdős task. + + +
+ main results +
+ +- On competitive programming (FrontierCS), Finch-9B averages 46.01 vs base Qwen3.5-9B's 32.46; on CALICO's P263 (UC Berkeley's official open-ended contest) it scores 86.10 vs 55.09 + + +
+ frontiercs results +
+ +- With preference learning (KTO), Finch-8B surpasses the best human score on AC1 and AC2, while its competitive programming score improves from 24.56 → 37.30. +- Finch-8B matches SOTA on two circle-packing tasks and improves the Erdős task by +3.2%. + +
+ frontiercs results +
+ +## Limitations + +Trajectories are collected and evaluated only with **OpenEvolve**; behavior under different scaffolds is not guaranteed. + +## License + +The **Finch Collection** is released under the [**CC-BY 4.0 License**](https://creativecommons.org/licenses/by/4.0/) and is recommended for **non-commercial academic research**. The accompanying **code** and **Finch model weights** are released under the [**Apache 2.0 License**](https://www.apache.org/licenses/LICENSE-2.0). + +## Acknowledgements + +This research was supported by the "Advanced GPU Utilization Support Program" funded by the Government of the Republic of Korea (Ministry of Science and ICT). We are grateful to the SkyDiscover team for their valuable feedback on the dataset construction process, the use of the SkyDiscover framework, and the overall direction of this research — in particular, [Shu Liu](https://shulynnliu.com/), [Shubham Agarwal](https://skejriwal44.github.io/), and [Mert Cemri](https://people.eecs.berkeley.edu/~mert_cemri/) for their insightful comments and discussions. We also thank the OpenEvolve team, especially Ritik Vijayvergiya and [Asankhaya Sharma](https://asankhaya.github.io/), for their guidance on using the OpenEvolve framework and for their thoughtful comments on this work. We further thank the authors of ALE-Bench, especially [Yuki Imajuku](https://imajuku.tech/), and the AtCoder team for authorizing the public release of the evolutionary search trajectories derived from their CC BY-ND 4.0-licensed dataset. Finally, we thank [Byung-Kwan Lee](https://byungkwanlee.github.io/ByungKwanLee-CV/) for valuable feedback during the early stages of this project. + + +## Citation + +```bibtex +@misc{lee2026evolutionfinetuninglearningdiscover, + title={Evolution Fine-Tuning: Learning to Discover Across 371 Optimization Tasks}, + author={Young-Jun Lee and Seungone Kim and Minki Kang and Alistair Cheong Liang Chuen and Zerui Chen and Seungho Han and Taehee Jung and Dongyeop Kang}, + year={2026}, + eprint={2606.29082}, + archivePrefix={arXiv}, + primaryClass={cs.CL}, + url={https://arxiv.org/abs/2606.29082}, +} +``` diff --git a/assets/finch_icon.png b/assets/finch_icon.png new file mode 100644 index 0000000..85a1f4e Binary files /dev/null and b/assets/finch_icon.png differ diff --git a/assets/intro_teaser.png b/assets/intro_teaser.png new file mode 100644 index 0000000..43ac633 --- /dev/null +++ b/assets/intro_teaser.png @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:50e825db841e72ac37560145a523697a4b009e320df28193f23bb467f9e8a85d +size 436148 diff --git a/assets/results_kto.png b/assets/results_kto.png new file mode 100644 index 0000000..57dc096 --- /dev/null +++ b/assets/results_kto.png @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b5da7421d1350a0fd86fcb649a9f000bf5420259d316c5a1898673db951295f8 +size 153143 diff --git a/chat_template.jinja b/chat_template.jinja new file mode 100644 index 0000000..01be9b3 --- /dev/null +++ b/chat_template.jinja @@ -0,0 +1,89 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0].role == 'system' %} + {{- messages[0].content + '\n\n' }} + {%- endif %} + {{- "# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0].role == 'system' %} + {{- '<|im_start|>system\n' + messages[0].content + '<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %} +{%- for message in messages[::-1] %} + {%- set index = (messages|length - 1) - loop.index0 %} + {%- if ns.multi_step_tool and message.role == "user" and message.content is string and not(message.content.startswith('') and message.content.endswith('')) %} + {%- set ns.multi_step_tool = false %} + {%- set ns.last_query_index = index %} + {%- endif %} +{%- endfor %} +{%- for message in messages %} + {%- if message.content is string %} + {%- set content = message.content %} + {%- else %} + {%- set content = '' %} + {%- endif %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) %} + {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {%- set reasoning_content = '' %} + {%- if message.reasoning_content is string %} + {%- set reasoning_content = message.reasoning_content %} + {%- else %} + {%- if '' in content %} + {%- set reasoning_content = content.split('')[0].rstrip('\n').split('')[-1].lstrip('\n') %} + {%- set content = content.split('')[-1].lstrip('\n') %} + {%- endif %} + {%- endif %} + {%- if loop.index0 > ns.last_query_index %} + {%- if loop.last or (not loop.last and reasoning_content) %} + {{- '<|im_start|>' + message.role + '\n\n' + reasoning_content.strip('\n') + '\n\n\n' + content.lstrip('\n') }} + {%- else %} + {{- '<|im_start|>' + message.role + '\n' + content }} + {%- endif %} + {%- else %} + {{- '<|im_start|>' + message.role + '\n' + content }} + {%- endif %} + {%- if message.tool_calls %} + {%- for tool_call in message.tool_calls %} + {%- if (loop.first and content) or (not loop.first) %} + {{- '\n' }} + {%- endif %} + {%- if tool_call.function %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {%- if tool_call.arguments is string %} + {{- tool_call.arguments }} + {%- else %} + {{- tool_call.arguments | tojson }} + {%- endif %} + {{- '}\n' }} + {%- endfor %} + {%- endif %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if loop.first or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} + {%- if enable_thinking is defined and enable_thinking is false %} + {{- '\n\n\n\n' }} + {%- endif %} +{%- endif %} \ No newline at end of file diff --git a/config.json b/config.json new file mode 100644 index 0000000..91e42ce --- /dev/null +++ b/config.json @@ -0,0 +1,74 @@ +{ + "architectures": [ + "Qwen3ForCausalLM" + ], + "attention_bias": false, + "attention_dropout": 0.0, + "bos_token_id": null, + "dtype": "bfloat16", + "eos_token_id": 151645, + "head_dim": 128, + "hidden_act": "silu", + "hidden_size": 4096, + "initializer_range": 0.02, + "intermediate_size": 12288, + "layer_types": [ + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention" + ], + "max_position_embeddings": 131072, + "max_window_layers": 36, + "model_type": "qwen3", + "num_attention_heads": 32, + "num_hidden_layers": 36, + "num_key_value_heads": 8, + "pad_token_id": 151643, + "rms_norm_eps": 1e-06, + "rope_parameters": { + "attention_factor": 1.0, + "factor": 1.5, + "original_max_position_embeddings": 131072, + "rope_theta": 1000000, + "rope_type": "yarn" + }, + "sliding_window": null, + "tie_word_embeddings": false, + "transformers_version": "5.2.0", + "use_cache": false, + "use_sliding_window": false, + "vocab_size": 151936 +} diff --git a/generation_config.json b/generation_config.json new file mode 100644 index 0000000..1701c94 --- /dev/null +++ b/generation_config.json @@ -0,0 +1,12 @@ +{ + "do_sample": true, + "eos_token_id": [ + 151645, + 151643 + ], + "pad_token_id": 151643, + "temperature": 0.6, + "top_k": 20, + "top_p": 0.95, + "transformers_version": "5.2.0" +} diff --git a/model.safetensors b/model.safetensors new file mode 100644 index 0000000..22f57c8 --- /dev/null +++ b/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7248a69f581975c50723005a85c27a3750e0e10d01d42012f4d83d215bea00c7 +size 16381517208 diff --git a/tokenizer.json b/tokenizer.json new file mode 100644 index 0000000..c7afbed --- /dev/null +++ b/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:be75606093db2094d7cd20f3c2f385c212750648bd6ea4fb2bf507a6a4c55506 +size 11422650 diff --git a/tokenizer_config.json b/tokenizer_config.json new file mode 100644 index 0000000..145e2c7 --- /dev/null +++ b/tokenizer_config.json @@ -0,0 +1,30 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|im_end|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "padding_side": "right", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +}