commit 42a8f751349d64217097b2c01ddc861588e19206 Author: ModelHub XC Date: Sun Sep 6 18:29:15 2026 +0800 初始化项目,由ModelHub XC社区提供模型 Model: Nirbhayhero07/deepsentinel-overseer-small Source: Original Platform diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..3fb2daf --- /dev/null +++ b/.gitattributes @@ -0,0 +1,39 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text +checkpoint-100/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-200/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-250/tokenizer.json filter=lfs diff=lfs merge=lfs -text +tokenizer.json filter=lfs diff=lfs merge=lfs -text diff --git a/README.md b/README.md new file mode 100644 index 0000000..89ddbb5 --- /dev/null +++ b/README.md @@ -0,0 +1,67 @@ +--- +base_model: Qwen/Qwen2.5-0.5B-Instruct +library_name: transformers +model_name: deepsentinel_model_small +tags: +- generated_from_trainer +- grpo +- trl +licence: license +--- + +# Model Card for deepsentinel_model_small + +This model is a fine-tuned version of [Qwen/Qwen2.5-0.5B-Instruct](https://huggingface.co/Qwen/Qwen2.5-0.5B-Instruct). +It has been trained using [TRL](https://github.com/huggingface/trl). + +## Quick start + +```python +from transformers import pipeline + +question = "If you had a time machine, but could only go to the past or the future once and never return, which would you choose and why?" +generator = pipeline("text-generation", model="None", device="cuda") +output = generator([{"role": "user", "content": question}], max_new_tokens=128, return_full_text=False)[0] +print(output["generated_text"]) +``` + +## Training procedure + + + + + +This model was trained with GRPO, a method introduced in [DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models](https://huggingface.co/papers/2402.03300). + +### Framework versions + +- TRL: 1.2.0 +- Transformers: 5.0.0 +- Pytorch: 2.10.0+cu128 +- Datasets: 4.8.4 +- Tokenizers: 0.22.2 + +## Citations + +Cite GRPO as: + +```bibtex +@article{shao2024deepseekmath, + title = {{DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models}}, + author = {Zhihong Shao and Peiyi Wang and Qihao Zhu and Runxin Xu and Junxiao Song and Mingchuan Zhang and Y. K. Li and Y. Wu and Daya Guo}, + year = 2024, + eprint = {arXiv:2402.03300}, +} +``` + +Cite TRL as: + +```bibtex +@software{vonwerra2020trl, + title = {{TRL: Transformers Reinforcement Learning}}, + author = {von Werra, Leandro and Belkada, Younes and Tunstall, Lewis and Beeching, Edward and Thrush, Tristan and Lambert, Nathan and Huang, Shengyi and Rasul, Kashif and Gallouédec, Quentin}, + license = {Apache-2.0}, + url = {https://github.com/huggingface/trl}, + year = {2020} +} +``` \ No newline at end of file diff --git a/artifacts/loss_plot.png b/artifacts/loss_plot.png new file mode 100644 index 0000000..f565528 Binary files /dev/null and b/artifacts/loss_plot.png differ diff --git a/artifacts/reward_plot.png b/artifacts/reward_plot.png new file mode 100644 index 0000000..25e65bd Binary files /dev/null and b/artifacts/reward_plot.png differ diff --git a/artifacts/train.log b/artifacts/train.log new file mode 100644 index 0000000..087160a --- /dev/null +++ b/artifacts/train.log @@ -0,0 +1,77 @@ +Warning: You are sending unauthenticated requests to the HF Hub. Please set a HF_TOKEN to enable higher rate limits and faster downloads. +`torch_dtype` is deprecated! Use `dtype` instead! +============================================================ +DeepSentinel GRPO Training +============================================================ + +[1] Collecting 120 training episodes... + Collected 120 episodes + Distribution: 40 easy, 40 medium, 40 hard + +[2] Evaluating heuristic baseline... + easy: avg_reward=-0.3681 + medium: avg_reward=0.0049 + hard: avg_reward=0.0833 + +[3] Loading model: Qwen/Qwen2.5-0.5B-Instruct + Loading weights: 0%| | 0/290 [00:00system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are Qwen, created by Alibaba Cloud. You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are Qwen, created by Alibaba Cloud. You are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-100/chat_template.jinja b/checkpoint-100/chat_template.jinja new file mode 100644 index 0000000..bdf7919 --- /dev/null +++ b/checkpoint-100/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are Qwen, created by Alibaba Cloud. You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are Qwen, created by Alibaba Cloud. You are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-100/config.json b/checkpoint-100/config.json new file mode 100644 index 0000000..c920646 --- /dev/null +++ b/checkpoint-100/config.json @@ -0,0 +1,57 @@ +{ + "architectures": [ + "Qwen2ForCausalLM" + ], + "attention_dropout": 0.0, + "bos_token_id": null, + "dtype": "bfloat16", + "eos_token_id": 151645, + "hidden_act": "silu", + "hidden_size": 896, + "initializer_range": 0.02, + "intermediate_size": 4864, + "layer_types": [ + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention" + ], + "max_position_embeddings": 32768, + "max_window_layers": 21, + "model_type": "qwen2", + "num_attention_heads": 14, + "num_hidden_layers": 24, + "num_key_value_heads": 2, + "pad_token_id": 151643, + "rms_norm_eps": 1e-06, + "rope_parameters": { + "rope_theta": 1000000.0, + "rope_type": "default" + }, + "sliding_window": null, + "tie_word_embeddings": true, + "transformers_version": "5.0.0", + "use_cache": false, + "use_sliding_window": false, + "vocab_size": 151936 +} diff --git a/checkpoint-100/generation_config.json b/checkpoint-100/generation_config.json new file mode 100644 index 0000000..d58659f --- /dev/null +++ b/checkpoint-100/generation_config.json @@ -0,0 +1,13 @@ +{ + "do_sample": true, + "eos_token_id": [ + 151645, + 151643 + ], + "pad_token_id": 151643, + "repetition_penalty": 1.1, + "temperature": 0.7, + "top_k": 20, + "top_p": 0.8, + "transformers_version": "5.0.0" +} diff --git a/checkpoint-100/model.safetensors b/checkpoint-100/model.safetensors new file mode 100644 index 0000000..eb4c432 --- /dev/null +++ b/checkpoint-100/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c8e93e7edb2fca3ee7153407aa68df9fc7958fcb4caf3a1343e1bc03a283184c +size 988097824 diff --git a/checkpoint-100/optimizer.pt b/checkpoint-100/optimizer.pt new file mode 100644 index 0000000..a7ad44c --- /dev/null +++ b/checkpoint-100/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:654dbb5e5c52340fd6bbbed4076701c00ac95503d50a32b13269c616b07431da +size 1976378699 diff --git a/checkpoint-100/rng_state.pth b/checkpoint-100/rng_state.pth new file mode 100644 index 0000000..9250ddf --- /dev/null +++ b/checkpoint-100/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4562c38db0d1b4734547153fdf17fd74b325082e8fc241be1d78988aac5fdc54 +size 14645 diff --git a/checkpoint-100/scheduler.pt b/checkpoint-100/scheduler.pt new file mode 100644 index 0000000..77e7c88 --- /dev/null +++ b/checkpoint-100/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2d19e9d246a3b522af1266d86db2c0a2723b502fbb6e56448bd47c5229c1fe90 +size 1465 diff --git a/checkpoint-100/tokenizer.json b/checkpoint-100/tokenizer.json new file mode 100644 index 0000000..34510ff --- /dev/null +++ b/checkpoint-100/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3fd169731d2cbde95e10bf356d66d5997fd885dd8dbb6fb4684da3f23b2585d8 +size 11421892 diff --git a/checkpoint-100/tokenizer_config.json b/checkpoint-100/tokenizer_config.json new file mode 100644 index 0000000..7d75d3b --- /dev/null +++ b/checkpoint-100/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|im_end|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-100/trainer_state.json b/checkpoint-100/trainer_state.json new file mode 100644 index 0000000..bdc568c --- /dev/null +++ b/checkpoint-100/trainer_state.json @@ -0,0 +1,304 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 3.3333333333333335, + "eval_steps": 500, + "global_step": 100, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.8625, + "completions/max_length": 379.3, + "completions/max_terminated_length": 57.8, + "completions/mean_length": 362.7625, + "completions/mean_terminated_length": 46.4125, + "completions/min_length": 318.1, + "completions/min_terminated_length": 38.1, + "entropy": 0.5937434114515782, + "epoch": 0.3333333333333333, + "frac_reward_zero_std": 0.8, + "grad_norm": 0.0, + "learning_rate": 1.9280000000000002e-05, + "loss": -0.017474760115146638, + "num_tokens": 61847.0, + "reward": -0.0987912505865097, + "reward_std": 0.9303328216075897, + "rewards/reward_fn/mean": -0.0987912505865097, + "rewards/reward_fn/std": 0.9303328156471252, + "step": 10, + "step_time": 31.724731107300066 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 1.0, + "completions/max_length": 400.0, + "completions/max_terminated_length": 0.0, + "completions/mean_length": 400.0, + "completions/mean_terminated_length": 0.0, + "completions/min_length": 400.0, + "completions/min_terminated_length": 0.0, + "entropy": 0.6593769532628357, + "epoch": 0.6666666666666666, + "frac_reward_zero_std": 1.0, + "grad_norm": 0.0, + "learning_rate": 1.8480000000000003e-05, + "loss": 0.0, + "num_tokens": 126843.0, + "reward": 0.048249998688697816, + "reward_std": 0.8191121600568294, + "rewards/reward_fn/mean": 0.048249998688697816, + "rewards/reward_fn/std": 0.8191121838986873, + "step": 20, + "step_time": 33.554111711299946 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.975, + "completions/max_length": 400.0, + "completions/max_terminated_length": 49.7, + "completions/mean_length": 396.2125, + "completions/mean_terminated_length": 49.7, + "completions/min_length": 369.7, + "completions/min_terminated_length": 49.7, + "entropy": 0.5578590292367153, + "epoch": 1.0, + "frac_reward_zero_std": 0.9, + "grad_norm": 1.8046875, + "learning_rate": 1.768e-05, + "loss": -0.005067326501011849, + "num_tokens": 191538.0, + "reward": 0.11049998998641967, + "reward_std": 0.711869715154171, + "rewards/reward_fn/mean": 0.11049998998641967, + "rewards/reward_fn/std": 0.7118697345256806, + "step": 30, + "step_time": 33.49379607419994 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 1.0, + "completions/max_length": 400.0, + "completions/max_terminated_length": 0.0, + "completions/mean_length": 400.0, + "completions/mean_terminated_length": 0.0, + "completions/min_length": 400.0, + "completions/min_terminated_length": 0.0, + "entropy": 0.004339599603554234, + "epoch": 1.3333333333333333, + "frac_reward_zero_std": 0.95, + "grad_norm": 0.0, + "learning_rate": 1.688e-05, + "loss": 0.0, + "num_tokens": 256536.0, + "reward": 0.07887499779462814, + "reward_std": 0.8938172444701195, + "rewards/reward_fn/mean": 0.07887499779462814, + "rewards/reward_fn/std": 0.8938172534108162, + "step": 40, + "step_time": 33.4393088197 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.9875, + "completions/max_length": 400.0, + "completions/max_terminated_length": 0.9, + "completions/mean_length": 395.1125, + "completions/mean_terminated_length": 0.9, + "completions/min_length": 360.9, + "completions/min_terminated_length": 0.9, + "entropy": 0.004430754791246727, + "epoch": 1.6666666666666665, + "frac_reward_zero_std": 0.975, + "grad_norm": 0.384765625, + "learning_rate": 1.6080000000000002e-05, + "loss": 0.0, + "num_tokens": 321071.0, + "reward": -0.16437500417232515, + "reward_std": 1.063568216562271, + "rewards/reward_fn/mean": -0.16437500417232515, + "rewards/reward_fn/std": 1.0635682106018067, + "step": 50, + "step_time": 33.5646041849001 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.9625, + "completions/max_length": 400.0, + "completions/max_terminated_length": 3.7, + "completions/mean_length": 385.4625, + "completions/mean_terminated_length": 3.7, + "completions/min_length": 283.7, + "completions/min_terminated_length": 3.7, + "entropy": 0.09348616125644185, + "epoch": 2.0, + "frac_reward_zero_std": 0.925, + "grad_norm": 0.0, + "learning_rate": 1.5280000000000003e-05, + "loss": 0.009755914658308029, + "num_tokens": 384804.0, + "reward": -0.16562500596046448, + "reward_std": 0.9360396310687065, + "rewards/reward_fn/mean": -0.16562500596046448, + "rewards/reward_fn/std": 0.9360396608710289, + "step": 60, + "step_time": 33.48749011520003 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.8625, + "completions/max_length": 400.0, + "completions/max_terminated_length": 44.3, + "completions/mean_length": 351.7625, + "completions/mean_terminated_length": 42.26666679382324, + "completions/min_length": 159.8, + "completions/min_terminated_length": 39.8, + "entropy": 0.6013902719132602, + "epoch": 2.3333333333333335, + "frac_reward_zero_std": 0.925, + "grad_norm": 2.40625, + "learning_rate": 1.448e-05, + "loss": -0.011668374389410019, + "num_tokens": 445825.0, + "reward": 0.08912500143051147, + "reward_std": 0.9901719689369202, + "rewards/reward_fn/mean": 0.08912500143051147, + "rewards/reward_fn/std": 0.9901719927787781, + "step": 70, + "step_time": 33.48351585430014 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.9, + "completions/max_length": 400.0, + "completions/max_terminated_length": 36.8, + "completions/mean_length": 365.4875, + "completions/mean_terminated_length": 34.75, + "completions/min_length": 192.7, + "completions/min_terminated_length": 32.7, + "entropy": 0.7647990029305219, + "epoch": 2.6666666666666665, + "frac_reward_zero_std": 0.975, + "grad_norm": 0.0, + "learning_rate": 1.3680000000000003e-05, + "loss": 0.0, + "num_tokens": 507944.0, + "reward": -0.14375001043081284, + "reward_std": 0.6796593680977822, + "rewards/reward_fn/mean": -0.14375001043081284, + "rewards/reward_fn/std": 0.6796593815088272, + "step": 80, + "step_time": 33.443731949500034 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.95, + "completions/max_length": 400.0, + "completions/max_terminated_length": 38.7, + "completions/mean_length": 386.45, + "completions/mean_terminated_length": 35.4, + "completions/min_length": 312.1, + "completions/min_terminated_length": 32.1, + "entropy": 0.817728553712368, + "epoch": 3.0, + "frac_reward_zero_std": 1.0, + "grad_norm": 0.0, + "learning_rate": 1.2880000000000002e-05, + "loss": 0.0, + "num_tokens": 571920.0, + "reward": -0.09100000560283661, + "reward_std": 0.7491292104125022, + "rewards/reward_fn/mean": -0.09100000560283661, + "rewards/reward_fn/std": 0.7491292074322701, + "step": 90, + "step_time": 33.4239154713001 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.95, + "completions/max_length": 400.0, + "completions/max_terminated_length": 52.1, + "completions/mean_length": 386.5125, + "completions/mean_terminated_length": 52.1, + "completions/min_length": 292.1, + "completions/min_terminated_length": 52.1, + "entropy": 0.7871452756226063, + "epoch": 3.3333333333333335, + "frac_reward_zero_std": 1.0, + "grad_norm": 0.0, + "learning_rate": 1.2080000000000001e-05, + "loss": 0.0, + "num_tokens": 635831.0, + "reward": -0.11849999725818634, + "reward_std": 0.9780218183994294, + "rewards/reward_fn/mean": -0.11849999725818634, + "rewards/reward_fn/std": 0.9780218482017518, + "step": 100, + "step_time": 33.44708331829988 + } + ], + "logging_steps": 10, + "max_steps": 250, + "num_input_tokens_seen": 635831, + "num_train_epochs": 9, + "save_steps": 100, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 0.0, + "train_batch_size": 2, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-100/training_args.bin b/checkpoint-100/training_args.bin new file mode 100644 index 0000000..bafd374 --- /dev/null +++ b/checkpoint-100/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0f0dd240a760a35edcf19bfedd721f31bea53099058c16d322d918d5f830b64a +size 7057 diff --git a/checkpoint-200/chat_template.jinja b/checkpoint-200/chat_template.jinja new file mode 100644 index 0000000..bdf7919 --- /dev/null +++ b/checkpoint-200/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are Qwen, created by Alibaba Cloud. You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are Qwen, created by Alibaba Cloud. You are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-200/config.json b/checkpoint-200/config.json new file mode 100644 index 0000000..c920646 --- /dev/null +++ b/checkpoint-200/config.json @@ -0,0 +1,57 @@ +{ + "architectures": [ + "Qwen2ForCausalLM" + ], + "attention_dropout": 0.0, + "bos_token_id": null, + "dtype": "bfloat16", + "eos_token_id": 151645, + "hidden_act": "silu", + "hidden_size": 896, + "initializer_range": 0.02, + "intermediate_size": 4864, + "layer_types": [ + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention" + ], + "max_position_embeddings": 32768, + "max_window_layers": 21, + "model_type": "qwen2", + "num_attention_heads": 14, + "num_hidden_layers": 24, + "num_key_value_heads": 2, + "pad_token_id": 151643, + "rms_norm_eps": 1e-06, + "rope_parameters": { + "rope_theta": 1000000.0, + "rope_type": "default" + }, + "sliding_window": null, + "tie_word_embeddings": true, + "transformers_version": "5.0.0", + "use_cache": false, + "use_sliding_window": false, + "vocab_size": 151936 +} diff --git a/checkpoint-200/generation_config.json b/checkpoint-200/generation_config.json new file mode 100644 index 0000000..d58659f --- /dev/null +++ b/checkpoint-200/generation_config.json @@ -0,0 +1,13 @@ +{ + "do_sample": true, + "eos_token_id": [ + 151645, + 151643 + ], + "pad_token_id": 151643, + "repetition_penalty": 1.1, + "temperature": 0.7, + "top_k": 20, + "top_p": 0.8, + "transformers_version": "5.0.0" +} diff --git a/checkpoint-200/model.safetensors b/checkpoint-200/model.safetensors new file mode 100644 index 0000000..280db21 --- /dev/null +++ b/checkpoint-200/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c84aba5165dfd69644cb6316f3b3375193b0efe4a622abb141bfab133699a48f +size 988097824 diff --git a/checkpoint-200/optimizer.pt b/checkpoint-200/optimizer.pt new file mode 100644 index 0000000..603372a --- /dev/null +++ b/checkpoint-200/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b6e0906cabe1f0726e51f18953c33912747a07a53f8c596be347f8bfd6de0319 +size 1976378699 diff --git a/checkpoint-200/rng_state.pth b/checkpoint-200/rng_state.pth new file mode 100644 index 0000000..fa8c43e --- /dev/null +++ b/checkpoint-200/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a48e54fd1ade1441bcc9a3da8003d35a0e81a05b41997eb4c78e53062bb68fda +size 14645 diff --git a/checkpoint-200/scheduler.pt b/checkpoint-200/scheduler.pt new file mode 100644 index 0000000..4c13763 --- /dev/null +++ b/checkpoint-200/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a23f05f4402bb7e3181df91e26f2c401cfdf935eaee7bfea5795007f352fa683 +size 1465 diff --git a/checkpoint-200/tokenizer.json b/checkpoint-200/tokenizer.json new file mode 100644 index 0000000..34510ff --- /dev/null +++ b/checkpoint-200/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3fd169731d2cbde95e10bf356d66d5997fd885dd8dbb6fb4684da3f23b2585d8 +size 11421892 diff --git a/checkpoint-200/tokenizer_config.json b/checkpoint-200/tokenizer_config.json new file mode 100644 index 0000000..7d75d3b --- /dev/null +++ b/checkpoint-200/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|im_end|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-200/trainer_state.json b/checkpoint-200/trainer_state.json new file mode 100644 index 0000000..45fea2f --- /dev/null +++ b/checkpoint-200/trainer_state.json @@ -0,0 +1,574 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 6.666666666666667, + "eval_steps": 500, + "global_step": 200, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.8625, + "completions/max_length": 379.3, + "completions/max_terminated_length": 57.8, + "completions/mean_length": 362.7625, + "completions/mean_terminated_length": 46.4125, + "completions/min_length": 318.1, + "completions/min_terminated_length": 38.1, + "entropy": 0.5937434114515782, + "epoch": 0.3333333333333333, + "frac_reward_zero_std": 0.8, + "grad_norm": 0.0, + "learning_rate": 1.9280000000000002e-05, + "loss": -0.017474760115146638, + "num_tokens": 61847.0, + "reward": -0.0987912505865097, + "reward_std": 0.9303328216075897, + "rewards/reward_fn/mean": -0.0987912505865097, + "rewards/reward_fn/std": 0.9303328156471252, + "step": 10, + "step_time": 31.724731107300066 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 1.0, + "completions/max_length": 400.0, + "completions/max_terminated_length": 0.0, + "completions/mean_length": 400.0, + "completions/mean_terminated_length": 0.0, + "completions/min_length": 400.0, + "completions/min_terminated_length": 0.0, + "entropy": 0.6593769532628357, + "epoch": 0.6666666666666666, + "frac_reward_zero_std": 1.0, + "grad_norm": 0.0, + "learning_rate": 1.8480000000000003e-05, + "loss": 0.0, + "num_tokens": 126843.0, + "reward": 0.048249998688697816, + "reward_std": 0.8191121600568294, + "rewards/reward_fn/mean": 0.048249998688697816, + "rewards/reward_fn/std": 0.8191121838986873, + "step": 20, + "step_time": 33.554111711299946 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.975, + "completions/max_length": 400.0, + "completions/max_terminated_length": 49.7, + "completions/mean_length": 396.2125, + "completions/mean_terminated_length": 49.7, + "completions/min_length": 369.7, + "completions/min_terminated_length": 49.7, + "entropy": 0.5578590292367153, + "epoch": 1.0, + "frac_reward_zero_std": 0.9, + "grad_norm": 1.8046875, + "learning_rate": 1.768e-05, + "loss": -0.005067326501011849, + "num_tokens": 191538.0, + "reward": 0.11049998998641967, + "reward_std": 0.711869715154171, + "rewards/reward_fn/mean": 0.11049998998641967, + "rewards/reward_fn/std": 0.7118697345256806, + "step": 30, + "step_time": 33.49379607419994 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 1.0, + "completions/max_length": 400.0, + "completions/max_terminated_length": 0.0, + "completions/mean_length": 400.0, + "completions/mean_terminated_length": 0.0, + "completions/min_length": 400.0, + "completions/min_terminated_length": 0.0, + "entropy": 0.004339599603554234, + "epoch": 1.3333333333333333, + "frac_reward_zero_std": 0.95, + "grad_norm": 0.0, + "learning_rate": 1.688e-05, + "loss": 0.0, + "num_tokens": 256536.0, + "reward": 0.07887499779462814, + "reward_std": 0.8938172444701195, + "rewards/reward_fn/mean": 0.07887499779462814, + "rewards/reward_fn/std": 0.8938172534108162, + "step": 40, + "step_time": 33.4393088197 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.9875, + "completions/max_length": 400.0, + "completions/max_terminated_length": 0.9, + "completions/mean_length": 395.1125, + "completions/mean_terminated_length": 0.9, + "completions/min_length": 360.9, + "completions/min_terminated_length": 0.9, + "entropy": 0.004430754791246727, + "epoch": 1.6666666666666665, + "frac_reward_zero_std": 0.975, + "grad_norm": 0.384765625, + "learning_rate": 1.6080000000000002e-05, + "loss": 0.0, + "num_tokens": 321071.0, + "reward": -0.16437500417232515, + "reward_std": 1.063568216562271, + "rewards/reward_fn/mean": -0.16437500417232515, + "rewards/reward_fn/std": 1.0635682106018067, + "step": 50, + "step_time": 33.5646041849001 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.9625, + "completions/max_length": 400.0, + "completions/max_terminated_length": 3.7, + "completions/mean_length": 385.4625, + "completions/mean_terminated_length": 3.7, + "completions/min_length": 283.7, + "completions/min_terminated_length": 3.7, + "entropy": 0.09348616125644185, + "epoch": 2.0, + "frac_reward_zero_std": 0.925, + "grad_norm": 0.0, + "learning_rate": 1.5280000000000003e-05, + "loss": 0.009755914658308029, + "num_tokens": 384804.0, + "reward": -0.16562500596046448, + "reward_std": 0.9360396310687065, + "rewards/reward_fn/mean": -0.16562500596046448, + "rewards/reward_fn/std": 0.9360396608710289, + "step": 60, + "step_time": 33.48749011520003 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.8625, + "completions/max_length": 400.0, + "completions/max_terminated_length": 44.3, + "completions/mean_length": 351.7625, + "completions/mean_terminated_length": 42.26666679382324, + "completions/min_length": 159.8, + "completions/min_terminated_length": 39.8, + "entropy": 0.6013902719132602, + "epoch": 2.3333333333333335, + "frac_reward_zero_std": 0.925, + "grad_norm": 2.40625, + "learning_rate": 1.448e-05, + "loss": -0.011668374389410019, + "num_tokens": 445825.0, + "reward": 0.08912500143051147, + "reward_std": 0.9901719689369202, + "rewards/reward_fn/mean": 0.08912500143051147, + "rewards/reward_fn/std": 0.9901719927787781, + "step": 70, + "step_time": 33.48351585430014 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.9, + "completions/max_length": 400.0, + "completions/max_terminated_length": 36.8, + "completions/mean_length": 365.4875, + "completions/mean_terminated_length": 34.75, + "completions/min_length": 192.7, + "completions/min_terminated_length": 32.7, + "entropy": 0.7647990029305219, + "epoch": 2.6666666666666665, + "frac_reward_zero_std": 0.975, + "grad_norm": 0.0, + "learning_rate": 1.3680000000000003e-05, + "loss": 0.0, + "num_tokens": 507944.0, + "reward": -0.14375001043081284, + "reward_std": 0.6796593680977822, + "rewards/reward_fn/mean": -0.14375001043081284, + "rewards/reward_fn/std": 0.6796593815088272, + "step": 80, + "step_time": 33.443731949500034 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.95, + "completions/max_length": 400.0, + "completions/max_terminated_length": 38.7, + "completions/mean_length": 386.45, + "completions/mean_terminated_length": 35.4, + "completions/min_length": 312.1, + "completions/min_terminated_length": 32.1, + "entropy": 0.817728553712368, + "epoch": 3.0, + "frac_reward_zero_std": 1.0, + "grad_norm": 0.0, + "learning_rate": 1.2880000000000002e-05, + "loss": 0.0, + "num_tokens": 571920.0, + "reward": -0.09100000560283661, + "reward_std": 0.7491292104125022, + "rewards/reward_fn/mean": -0.09100000560283661, + "rewards/reward_fn/std": 0.7491292074322701, + "step": 90, + "step_time": 33.4239154713001 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.95, + "completions/max_length": 400.0, + "completions/max_terminated_length": 52.1, + "completions/mean_length": 386.5125, + "completions/mean_terminated_length": 52.1, + "completions/min_length": 292.1, + "completions/min_terminated_length": 52.1, + "entropy": 0.7871452756226063, + "epoch": 3.3333333333333335, + "frac_reward_zero_std": 1.0, + "grad_norm": 0.0, + "learning_rate": 1.2080000000000001e-05, + "loss": 0.0, + "num_tokens": 635831.0, + "reward": -0.11849999725818634, + "reward_std": 0.9780218183994294, + "rewards/reward_fn/mean": -0.11849999725818634, + "rewards/reward_fn/std": 0.9780218482017518, + "step": 100, + "step_time": 33.44708331829988 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.95, + "completions/max_length": 400.0, + "completions/max_terminated_length": 75.3, + "completions/mean_length": 389.775, + "completions/mean_terminated_length": 65.95, + "completions/min_length": 336.6, + "completions/min_terminated_length": 56.6, + "entropy": 0.8245016686618328, + "epoch": 3.6666666666666665, + "frac_reward_zero_std": 1.0, + "grad_norm": 0.0, + "learning_rate": 1.128e-05, + "loss": 0.0, + "num_tokens": 699917.0, + "reward": 0.06824999898672104, + "reward_std": 0.8597677208483219, + "rewards/reward_fn/mean": 0.06824999898672104, + "rewards/reward_fn/std": 0.8597677327692509, + "step": 110, + "step_time": 33.43369797569976 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.9875, + "completions/max_length": 400.0, + "completions/max_terminated_length": 29.0, + "completions/mean_length": 398.625, + "completions/mean_terminated_length": 29.0, + "completions/min_length": 389.0, + "completions/min_terminated_length": 29.0, + "entropy": 0.8096485503017903, + "epoch": 4.0, + "frac_reward_zero_std": 1.0, + "grad_norm": 0.0, + "learning_rate": 1.0480000000000001e-05, + "loss": 0.0, + "num_tokens": 764733.0, + "reward": -0.05875001698732376, + "reward_std": 0.7073742881417274, + "rewards/reward_fn/mean": -0.05875001698732376, + "rewards/reward_fn/std": 0.7073742985725403, + "step": 120, + "step_time": 33.47654410239984 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.9625, + "completions/max_length": 400.0, + "completions/max_terminated_length": 51.8, + "completions/mean_length": 391.475, + "completions/mean_terminated_length": 51.8, + "completions/min_length": 331.8, + "completions/min_terminated_length": 51.8, + "entropy": 0.8178432904183864, + "epoch": 4.333333333333333, + "frac_reward_zero_std": 0.975, + "grad_norm": 0.0, + "learning_rate": 9.68e-06, + "loss": 0.0, + "num_tokens": 828855.0, + "reward": -0.017000006139278413, + "reward_std": 0.7560826383531094, + "rewards/reward_fn/mean": -0.017000006139278413, + "rewards/reward_fn/std": 0.7560826435685157, + "step": 130, + "step_time": 33.514995820599964 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.9375, + "completions/max_length": 400.0, + "completions/max_terminated_length": 46.2, + "completions/mean_length": 380.775, + "completions/mean_terminated_length": 46.2, + "completions/min_length": 246.2, + "completions/min_terminated_length": 46.2, + "entropy": 0.8854692216962576, + "epoch": 4.666666666666667, + "frac_reward_zero_std": 0.95, + "grad_norm": 0.0, + "learning_rate": 8.880000000000001e-06, + "loss": -0.012354153394699096, + "num_tokens": 892373.0, + "reward": -0.13475000411272048, + "reward_std": 0.9627669990062714, + "rewards/reward_fn/mean": -0.13475000411272048, + "rewards/reward_fn/std": 0.9627670347690582, + "step": 140, + "step_time": 33.50577180429955 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.975, + "completions/max_length": 400.0, + "completions/max_terminated_length": 32.3, + "completions/mean_length": 394.275, + "completions/mean_terminated_length": 17.1, + "completions/min_length": 361.9, + "completions/min_terminated_length": 1.9, + "entropy": 0.8237141810357571, + "epoch": 5.0, + "frac_reward_zero_std": 1.0, + "grad_norm": 0.0, + "learning_rate": 8.08e-06, + "loss": 0.0, + "num_tokens": 956875.0, + "reward": -0.030000001192092896, + "reward_std": 0.7297740295529366, + "rewards/reward_fn/mean": -0.030000001192092896, + "rewards/reward_fn/std": 0.7297740370035172, + "step": 150, + "step_time": 33.46883516300004 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.975, + "completions/max_length": 400.0, + "completions/max_terminated_length": 38.5, + "completions/mean_length": 394.8125, + "completions/mean_terminated_length": 38.5, + "completions/min_length": 358.5, + "completions/min_terminated_length": 38.5, + "entropy": 0.9815139673650265, + "epoch": 5.333333333333333, + "frac_reward_zero_std": 1.0, + "grad_norm": 0.0, + "learning_rate": 7.280000000000001e-06, + "loss": 0.0, + "num_tokens": 1021402.0, + "reward": -0.06250000596046448, + "reward_std": 0.9604951858520507, + "rewards/reward_fn/mean": -0.06250000596046448, + "rewards/reward_fn/std": 0.9604952037334442, + "step": 160, + "step_time": 33.41488298210034 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.975, + "completions/max_length": 400.0, + "completions/max_terminated_length": 21.2, + "completions/mean_length": 392.65, + "completions/mean_terminated_length": 21.2, + "completions/min_length": 341.2, + "completions/min_terminated_length": 21.2, + "entropy": 0.9070465706288815, + "epoch": 5.666666666666667, + "frac_reward_zero_std": 0.975, + "grad_norm": 0.0, + "learning_rate": 6.480000000000001e-06, + "loss": 0.0, + "num_tokens": 1085632.0, + "reward": -0.1513125091791153, + "reward_std": 0.855255764722824, + "rewards/reward_fn/mean": -0.1513125091791153, + "rewards/reward_fn/std": 0.8552557826042175, + "step": 170, + "step_time": 33.4777201513999 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 1.0, + "completions/max_length": 400.0, + "completions/max_terminated_length": 0.0, + "completions/mean_length": 400.0, + "completions/mean_terminated_length": 0.0, + "completions/min_length": 400.0, + "completions/min_terminated_length": 0.0, + "entropy": 0.9226562466472388, + "epoch": 6.0, + "frac_reward_zero_std": 1.0, + "grad_norm": 0.0, + "learning_rate": 5.68e-06, + "loss": 0.0, + "num_tokens": 1150692.0, + "reward": 0.08200000673532486, + "reward_std": 0.9346169531345367, + "rewards/reward_fn/mean": 0.08200000673532486, + "rewards/reward_fn/std": 0.9346169650554657, + "step": 180, + "step_time": 33.47372673979989 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.975, + "completions/max_length": 400.0, + "completions/max_terminated_length": 35.5, + "completions/mean_length": 394.4375, + "completions/mean_terminated_length": 35.5, + "completions/min_length": 355.5, + "completions/min_terminated_length": 35.5, + "entropy": 0.9981254413723946, + "epoch": 6.333333333333333, + "frac_reward_zero_std": 0.975, + "grad_norm": 0.0, + "learning_rate": 4.880000000000001e-06, + "loss": 0.0, + "num_tokens": 1215133.0, + "reward": -0.08775000870227814, + "reward_std": 0.8883017361164093, + "rewards/reward_fn/mean": -0.08775000870227814, + "rewards/reward_fn/std": 0.8883017897605896, + "step": 190, + "step_time": 33.41868621859985 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.9875, + "completions/max_length": 400.0, + "completions/max_terminated_length": 30.4, + "completions/mean_length": 398.8, + "completions/mean_terminated_length": 30.4, + "completions/min_length": 390.4, + "completions/min_terminated_length": 30.4, + "entropy": 0.9789559677243233, + "epoch": 6.666666666666667, + "frac_reward_zero_std": 1.0, + "grad_norm": 0.0, + "learning_rate": 4.08e-06, + "loss": 0.0, + "num_tokens": 1279879.0, + "reward": -0.03525000065565109, + "reward_std": 0.8935358546674251, + "rewards/reward_fn/mean": -0.03525000065565109, + "rewards/reward_fn/std": 0.8935358554124833, + "step": 200, + "step_time": 33.424166100800086 + } + ], + "logging_steps": 10, + "max_steps": 250, + "num_input_tokens_seen": 1279879, + "num_train_epochs": 9, + "save_steps": 100, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 0.0, + "train_batch_size": 2, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-200/training_args.bin b/checkpoint-200/training_args.bin new file mode 100644 index 0000000..bafd374 --- /dev/null +++ b/checkpoint-200/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0f0dd240a760a35edcf19bfedd721f31bea53099058c16d322d918d5f830b64a +size 7057 diff --git a/checkpoint-250/chat_template.jinja b/checkpoint-250/chat_template.jinja new file mode 100644 index 0000000..bdf7919 --- /dev/null +++ b/checkpoint-250/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are Qwen, created by Alibaba Cloud. You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are Qwen, created by Alibaba Cloud. You are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-250/config.json b/checkpoint-250/config.json new file mode 100644 index 0000000..c920646 --- /dev/null +++ b/checkpoint-250/config.json @@ -0,0 +1,57 @@ +{ + "architectures": [ + "Qwen2ForCausalLM" + ], + "attention_dropout": 0.0, + "bos_token_id": null, + "dtype": "bfloat16", + "eos_token_id": 151645, + "hidden_act": "silu", + "hidden_size": 896, + "initializer_range": 0.02, + "intermediate_size": 4864, + "layer_types": [ + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention" + ], + "max_position_embeddings": 32768, + "max_window_layers": 21, + "model_type": "qwen2", + "num_attention_heads": 14, + "num_hidden_layers": 24, + "num_key_value_heads": 2, + "pad_token_id": 151643, + "rms_norm_eps": 1e-06, + "rope_parameters": { + "rope_theta": 1000000.0, + "rope_type": "default" + }, + "sliding_window": null, + "tie_word_embeddings": true, + "transformers_version": "5.0.0", + "use_cache": false, + "use_sliding_window": false, + "vocab_size": 151936 +} diff --git a/checkpoint-250/generation_config.json b/checkpoint-250/generation_config.json new file mode 100644 index 0000000..d58659f --- /dev/null +++ b/checkpoint-250/generation_config.json @@ -0,0 +1,13 @@ +{ + "do_sample": true, + "eos_token_id": [ + 151645, + 151643 + ], + "pad_token_id": 151643, + "repetition_penalty": 1.1, + "temperature": 0.7, + "top_k": 20, + "top_p": 0.8, + "transformers_version": "5.0.0" +} diff --git a/checkpoint-250/model.safetensors b/checkpoint-250/model.safetensors new file mode 100644 index 0000000..76221d2 --- /dev/null +++ b/checkpoint-250/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2f6ecf368a5199e23b82403ab68284fec38fb1f66520d261e9e8cb1f6b4ede31 +size 988097824 diff --git a/checkpoint-250/optimizer.pt b/checkpoint-250/optimizer.pt new file mode 100644 index 0000000..12e76a9 --- /dev/null +++ b/checkpoint-250/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2d8ddb2a1eae3170886a27db3b5ac9bf3aedea6f613bc1f3564d112947682900 +size 1976378699 diff --git a/checkpoint-250/rng_state.pth b/checkpoint-250/rng_state.pth new file mode 100644 index 0000000..48234c3 --- /dev/null +++ b/checkpoint-250/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:979e24915498a0c3479f76d4e80c72ae05419a1971ce8f05dcfdb1d82b4c524f +size 14645 diff --git a/checkpoint-250/scheduler.pt b/checkpoint-250/scheduler.pt new file mode 100644 index 0000000..811ded8 --- /dev/null +++ b/checkpoint-250/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5b04611bea80597dee677a377375d057cf3fa3437fb4ed8809d4662285b8811b +size 1465 diff --git a/checkpoint-250/tokenizer.json b/checkpoint-250/tokenizer.json new file mode 100644 index 0000000..34510ff --- /dev/null +++ b/checkpoint-250/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3fd169731d2cbde95e10bf356d66d5997fd885dd8dbb6fb4684da3f23b2585d8 +size 11421892 diff --git a/checkpoint-250/tokenizer_config.json b/checkpoint-250/tokenizer_config.json new file mode 100644 index 0000000..7d75d3b --- /dev/null +++ b/checkpoint-250/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|im_end|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-250/trainer_state.json b/checkpoint-250/trainer_state.json new file mode 100644 index 0000000..a4a59c0 --- /dev/null +++ b/checkpoint-250/trainer_state.json @@ -0,0 +1,709 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 8.333333333333334, + "eval_steps": 500, + "global_step": 250, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.8625, + "completions/max_length": 379.3, + "completions/max_terminated_length": 57.8, + "completions/mean_length": 362.7625, + "completions/mean_terminated_length": 46.4125, + "completions/min_length": 318.1, + "completions/min_terminated_length": 38.1, + "entropy": 0.5937434114515782, + "epoch": 0.3333333333333333, + "frac_reward_zero_std": 0.8, + "grad_norm": 0.0, + "learning_rate": 1.9280000000000002e-05, + "loss": -0.017474760115146638, + "num_tokens": 61847.0, + "reward": -0.0987912505865097, + "reward_std": 0.9303328216075897, + "rewards/reward_fn/mean": -0.0987912505865097, + "rewards/reward_fn/std": 0.9303328156471252, + "step": 10, + "step_time": 31.724731107300066 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 1.0, + "completions/max_length": 400.0, + "completions/max_terminated_length": 0.0, + "completions/mean_length": 400.0, + "completions/mean_terminated_length": 0.0, + "completions/min_length": 400.0, + "completions/min_terminated_length": 0.0, + "entropy": 0.6593769532628357, + "epoch": 0.6666666666666666, + "frac_reward_zero_std": 1.0, + "grad_norm": 0.0, + "learning_rate": 1.8480000000000003e-05, + "loss": 0.0, + "num_tokens": 126843.0, + "reward": 0.048249998688697816, + "reward_std": 0.8191121600568294, + "rewards/reward_fn/mean": 0.048249998688697816, + "rewards/reward_fn/std": 0.8191121838986873, + "step": 20, + "step_time": 33.554111711299946 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.975, + "completions/max_length": 400.0, + "completions/max_terminated_length": 49.7, + "completions/mean_length": 396.2125, + "completions/mean_terminated_length": 49.7, + "completions/min_length": 369.7, + "completions/min_terminated_length": 49.7, + "entropy": 0.5578590292367153, + "epoch": 1.0, + "frac_reward_zero_std": 0.9, + "grad_norm": 1.8046875, + "learning_rate": 1.768e-05, + "loss": -0.005067326501011849, + "num_tokens": 191538.0, + "reward": 0.11049998998641967, + "reward_std": 0.711869715154171, + "rewards/reward_fn/mean": 0.11049998998641967, + "rewards/reward_fn/std": 0.7118697345256806, + "step": 30, + "step_time": 33.49379607419994 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 1.0, + "completions/max_length": 400.0, + "completions/max_terminated_length": 0.0, + "completions/mean_length": 400.0, + "completions/mean_terminated_length": 0.0, + "completions/min_length": 400.0, + "completions/min_terminated_length": 0.0, + "entropy": 0.004339599603554234, + "epoch": 1.3333333333333333, + "frac_reward_zero_std": 0.95, + "grad_norm": 0.0, + "learning_rate": 1.688e-05, + "loss": 0.0, + "num_tokens": 256536.0, + "reward": 0.07887499779462814, + "reward_std": 0.8938172444701195, + "rewards/reward_fn/mean": 0.07887499779462814, + "rewards/reward_fn/std": 0.8938172534108162, + "step": 40, + "step_time": 33.4393088197 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.9875, + "completions/max_length": 400.0, + "completions/max_terminated_length": 0.9, + "completions/mean_length": 395.1125, + "completions/mean_terminated_length": 0.9, + "completions/min_length": 360.9, + "completions/min_terminated_length": 0.9, + "entropy": 0.004430754791246727, + "epoch": 1.6666666666666665, + "frac_reward_zero_std": 0.975, + "grad_norm": 0.384765625, + "learning_rate": 1.6080000000000002e-05, + "loss": 0.0, + "num_tokens": 321071.0, + "reward": -0.16437500417232515, + "reward_std": 1.063568216562271, + "rewards/reward_fn/mean": -0.16437500417232515, + "rewards/reward_fn/std": 1.0635682106018067, + "step": 50, + "step_time": 33.5646041849001 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.9625, + "completions/max_length": 400.0, + "completions/max_terminated_length": 3.7, + "completions/mean_length": 385.4625, + "completions/mean_terminated_length": 3.7, + "completions/min_length": 283.7, + "completions/min_terminated_length": 3.7, + "entropy": 0.09348616125644185, + "epoch": 2.0, + "frac_reward_zero_std": 0.925, + "grad_norm": 0.0, + "learning_rate": 1.5280000000000003e-05, + "loss": 0.009755914658308029, + "num_tokens": 384804.0, + "reward": -0.16562500596046448, + "reward_std": 0.9360396310687065, + "rewards/reward_fn/mean": -0.16562500596046448, + "rewards/reward_fn/std": 0.9360396608710289, + "step": 60, + "step_time": 33.48749011520003 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.8625, + "completions/max_length": 400.0, + "completions/max_terminated_length": 44.3, + "completions/mean_length": 351.7625, + "completions/mean_terminated_length": 42.26666679382324, + "completions/min_length": 159.8, + "completions/min_terminated_length": 39.8, + "entropy": 0.6013902719132602, + "epoch": 2.3333333333333335, + "frac_reward_zero_std": 0.925, + "grad_norm": 2.40625, + "learning_rate": 1.448e-05, + "loss": -0.011668374389410019, + "num_tokens": 445825.0, + "reward": 0.08912500143051147, + "reward_std": 0.9901719689369202, + "rewards/reward_fn/mean": 0.08912500143051147, + "rewards/reward_fn/std": 0.9901719927787781, + "step": 70, + "step_time": 33.48351585430014 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.9, + "completions/max_length": 400.0, + "completions/max_terminated_length": 36.8, + "completions/mean_length": 365.4875, + "completions/mean_terminated_length": 34.75, + "completions/min_length": 192.7, + "completions/min_terminated_length": 32.7, + "entropy": 0.7647990029305219, + "epoch": 2.6666666666666665, + "frac_reward_zero_std": 0.975, + "grad_norm": 0.0, + "learning_rate": 1.3680000000000003e-05, + "loss": 0.0, + "num_tokens": 507944.0, + "reward": -0.14375001043081284, + "reward_std": 0.6796593680977822, + "rewards/reward_fn/mean": -0.14375001043081284, + "rewards/reward_fn/std": 0.6796593815088272, + "step": 80, + "step_time": 33.443731949500034 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.95, + "completions/max_length": 400.0, + "completions/max_terminated_length": 38.7, + "completions/mean_length": 386.45, + "completions/mean_terminated_length": 35.4, + "completions/min_length": 312.1, + "completions/min_terminated_length": 32.1, + "entropy": 0.817728553712368, + "epoch": 3.0, + "frac_reward_zero_std": 1.0, + "grad_norm": 0.0, + "learning_rate": 1.2880000000000002e-05, + "loss": 0.0, + "num_tokens": 571920.0, + "reward": -0.09100000560283661, + "reward_std": 0.7491292104125022, + "rewards/reward_fn/mean": -0.09100000560283661, + "rewards/reward_fn/std": 0.7491292074322701, + "step": 90, + "step_time": 33.4239154713001 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.95, + "completions/max_length": 400.0, + "completions/max_terminated_length": 52.1, + "completions/mean_length": 386.5125, + "completions/mean_terminated_length": 52.1, + "completions/min_length": 292.1, + "completions/min_terminated_length": 52.1, + "entropy": 0.7871452756226063, + "epoch": 3.3333333333333335, + "frac_reward_zero_std": 1.0, + "grad_norm": 0.0, + "learning_rate": 1.2080000000000001e-05, + "loss": 0.0, + "num_tokens": 635831.0, + "reward": -0.11849999725818634, + "reward_std": 0.9780218183994294, + "rewards/reward_fn/mean": -0.11849999725818634, + "rewards/reward_fn/std": 0.9780218482017518, + "step": 100, + "step_time": 33.44708331829988 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.95, + "completions/max_length": 400.0, + "completions/max_terminated_length": 75.3, + "completions/mean_length": 389.775, + "completions/mean_terminated_length": 65.95, + "completions/min_length": 336.6, + "completions/min_terminated_length": 56.6, + "entropy": 0.8245016686618328, + "epoch": 3.6666666666666665, + "frac_reward_zero_std": 1.0, + "grad_norm": 0.0, + "learning_rate": 1.128e-05, + "loss": 0.0, + "num_tokens": 699917.0, + "reward": 0.06824999898672104, + "reward_std": 0.8597677208483219, + "rewards/reward_fn/mean": 0.06824999898672104, + "rewards/reward_fn/std": 0.8597677327692509, + "step": 110, + "step_time": 33.43369797569976 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.9875, + "completions/max_length": 400.0, + "completions/max_terminated_length": 29.0, + "completions/mean_length": 398.625, + "completions/mean_terminated_length": 29.0, + "completions/min_length": 389.0, + "completions/min_terminated_length": 29.0, + "entropy": 0.8096485503017903, + "epoch": 4.0, + "frac_reward_zero_std": 1.0, + "grad_norm": 0.0, + "learning_rate": 1.0480000000000001e-05, + "loss": 0.0, + "num_tokens": 764733.0, + "reward": -0.05875001698732376, + "reward_std": 0.7073742881417274, + "rewards/reward_fn/mean": -0.05875001698732376, + "rewards/reward_fn/std": 0.7073742985725403, + "step": 120, + "step_time": 33.47654410239984 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.9625, + "completions/max_length": 400.0, + "completions/max_terminated_length": 51.8, + "completions/mean_length": 391.475, + "completions/mean_terminated_length": 51.8, + "completions/min_length": 331.8, + "completions/min_terminated_length": 51.8, + "entropy": 0.8178432904183864, + "epoch": 4.333333333333333, + "frac_reward_zero_std": 0.975, + "grad_norm": 0.0, + "learning_rate": 9.68e-06, + "loss": 0.0, + "num_tokens": 828855.0, + "reward": -0.017000006139278413, + "reward_std": 0.7560826383531094, + "rewards/reward_fn/mean": -0.017000006139278413, + "rewards/reward_fn/std": 0.7560826435685157, + "step": 130, + "step_time": 33.514995820599964 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.9375, + "completions/max_length": 400.0, + "completions/max_terminated_length": 46.2, + "completions/mean_length": 380.775, + "completions/mean_terminated_length": 46.2, + "completions/min_length": 246.2, + "completions/min_terminated_length": 46.2, + "entropy": 0.8854692216962576, + "epoch": 4.666666666666667, + "frac_reward_zero_std": 0.95, + "grad_norm": 0.0, + "learning_rate": 8.880000000000001e-06, + "loss": -0.012354153394699096, + "num_tokens": 892373.0, + "reward": -0.13475000411272048, + "reward_std": 0.9627669990062714, + "rewards/reward_fn/mean": -0.13475000411272048, + "rewards/reward_fn/std": 0.9627670347690582, + "step": 140, + "step_time": 33.50577180429955 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.975, + "completions/max_length": 400.0, + "completions/max_terminated_length": 32.3, + "completions/mean_length": 394.275, + "completions/mean_terminated_length": 17.1, + "completions/min_length": 361.9, + "completions/min_terminated_length": 1.9, + "entropy": 0.8237141810357571, + "epoch": 5.0, + "frac_reward_zero_std": 1.0, + "grad_norm": 0.0, + "learning_rate": 8.08e-06, + "loss": 0.0, + "num_tokens": 956875.0, + "reward": -0.030000001192092896, + "reward_std": 0.7297740295529366, + "rewards/reward_fn/mean": -0.030000001192092896, + "rewards/reward_fn/std": 0.7297740370035172, + "step": 150, + "step_time": 33.46883516300004 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.975, + "completions/max_length": 400.0, + "completions/max_terminated_length": 38.5, + "completions/mean_length": 394.8125, + "completions/mean_terminated_length": 38.5, + "completions/min_length": 358.5, + "completions/min_terminated_length": 38.5, + "entropy": 0.9815139673650265, + "epoch": 5.333333333333333, + "frac_reward_zero_std": 1.0, + "grad_norm": 0.0, + "learning_rate": 7.280000000000001e-06, + "loss": 0.0, + "num_tokens": 1021402.0, + "reward": -0.06250000596046448, + "reward_std": 0.9604951858520507, + "rewards/reward_fn/mean": -0.06250000596046448, + "rewards/reward_fn/std": 0.9604952037334442, + "step": 160, + "step_time": 33.41488298210034 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.975, + "completions/max_length": 400.0, + "completions/max_terminated_length": 21.2, + "completions/mean_length": 392.65, + "completions/mean_terminated_length": 21.2, + "completions/min_length": 341.2, + "completions/min_terminated_length": 21.2, + "entropy": 0.9070465706288815, + "epoch": 5.666666666666667, + "frac_reward_zero_std": 0.975, + "grad_norm": 0.0, + "learning_rate": 6.480000000000001e-06, + "loss": 0.0, + "num_tokens": 1085632.0, + "reward": -0.1513125091791153, + "reward_std": 0.855255764722824, + "rewards/reward_fn/mean": -0.1513125091791153, + "rewards/reward_fn/std": 0.8552557826042175, + "step": 170, + "step_time": 33.4777201513999 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 1.0, + "completions/max_length": 400.0, + "completions/max_terminated_length": 0.0, + "completions/mean_length": 400.0, + "completions/mean_terminated_length": 0.0, + "completions/min_length": 400.0, + "completions/min_terminated_length": 0.0, + "entropy": 0.9226562466472388, + "epoch": 6.0, + "frac_reward_zero_std": 1.0, + "grad_norm": 0.0, + "learning_rate": 5.68e-06, + "loss": 0.0, + "num_tokens": 1150692.0, + "reward": 0.08200000673532486, + "reward_std": 0.9346169531345367, + "rewards/reward_fn/mean": 0.08200000673532486, + "rewards/reward_fn/std": 0.9346169650554657, + "step": 180, + "step_time": 33.47372673979989 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.975, + "completions/max_length": 400.0, + "completions/max_terminated_length": 35.5, + "completions/mean_length": 394.4375, + "completions/mean_terminated_length": 35.5, + "completions/min_length": 355.5, + "completions/min_terminated_length": 35.5, + "entropy": 0.9981254413723946, + "epoch": 6.333333333333333, + "frac_reward_zero_std": 0.975, + "grad_norm": 0.0, + "learning_rate": 4.880000000000001e-06, + "loss": 0.0, + "num_tokens": 1215133.0, + "reward": -0.08775000870227814, + "reward_std": 0.8883017361164093, + "rewards/reward_fn/mean": -0.08775000870227814, + "rewards/reward_fn/std": 0.8883017897605896, + "step": 190, + "step_time": 33.41868621859985 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.9875, + "completions/max_length": 400.0, + "completions/max_terminated_length": 30.4, + "completions/mean_length": 398.8, + "completions/mean_terminated_length": 30.4, + "completions/min_length": 390.4, + "completions/min_terminated_length": 30.4, + "entropy": 0.9789559677243233, + "epoch": 6.666666666666667, + "frac_reward_zero_std": 1.0, + "grad_norm": 0.0, + "learning_rate": 4.08e-06, + "loss": 0.0, + "num_tokens": 1279879.0, + "reward": -0.03525000065565109, + "reward_std": 0.8935358546674251, + "rewards/reward_fn/mean": -0.03525000065565109, + "rewards/reward_fn/std": 0.8935358554124833, + "step": 200, + "step_time": 33.424166100800086 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.9625, + "completions/max_length": 400.0, + "completions/max_terminated_length": 57.2, + "completions/mean_length": 392.15, + "completions/mean_terminated_length": 57.2, + "completions/min_length": 337.2, + "completions/min_terminated_length": 57.2, + "entropy": 0.9129896691069007, + "epoch": 7.0, + "frac_reward_zero_std": 1.0, + "grad_norm": 0.0, + "learning_rate": 3.2800000000000004e-06, + "loss": 0.0, + "num_tokens": 1344343.0, + "reward": -0.011500009894371032, + "reward_std": 0.8285948574543, + "rewards/reward_fn/mean": -0.011500009894371032, + "rewards/reward_fn/std": 0.8285948634147644, + "step": 210, + "step_time": 33.59373969100034 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 1.0, + "completions/max_length": 400.0, + "completions/max_terminated_length": 0.0, + "completions/mean_length": 400.0, + "completions/mean_terminated_length": 0.0, + "completions/min_length": 400.0, + "completions/min_terminated_length": 0.0, + "entropy": 0.960437498241663, + "epoch": 7.333333333333333, + "frac_reward_zero_std": 1.0, + "grad_norm": 0.0, + "learning_rate": 2.4800000000000004e-06, + "loss": 0.0, + "num_tokens": 1409195.0, + "reward": 0.018250004947185518, + "reward_std": 0.8383742928504944, + "rewards/reward_fn/mean": 0.018250004947185518, + "rewards/reward_fn/std": 0.8383742868900299, + "step": 220, + "step_time": 33.32045125410023 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.9875, + "completions/max_length": 400.0, + "completions/max_terminated_length": 19.6, + "completions/mean_length": 397.45, + "completions/mean_terminated_length": 19.6, + "completions/min_length": 379.6, + "completions/min_terminated_length": 19.6, + "entropy": 0.9672328025102616, + "epoch": 7.666666666666667, + "frac_reward_zero_std": 1.0, + "grad_norm": 0.0, + "learning_rate": 1.6800000000000002e-06, + "loss": 0.0, + "num_tokens": 1473919.0, + "reward": -0.11225000917911529, + "reward_std": 0.8593939155340194, + "rewards/reward_fn/mean": -0.11225000917911529, + "rewards/reward_fn/std": 0.8593939125537873, + "step": 230, + "step_time": 33.41978766690026 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.9625, + "completions/max_length": 400.0, + "completions/max_terminated_length": 36.1, + "completions/mean_length": 389.5125, + "completions/mean_terminated_length": 36.1, + "completions/min_length": 316.1, + "completions/min_terminated_length": 36.1, + "entropy": 1.013204599916935, + "epoch": 8.0, + "frac_reward_zero_std": 1.0, + "grad_norm": 0.0, + "learning_rate": 8.8e-07, + "loss": 0.0, + "num_tokens": 1538120.0, + "reward": -0.015000000596046448, + "reward_std": 0.9198581218719483, + "rewards/reward_fn/mean": -0.015000000596046448, + "rewards/reward_fn/std": 0.9198581159114838, + "step": 240, + "step_time": 33.588851060100204 + }, + { + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/clipped_ratio": 0.9625, + "completions/max_length": 400.0, + "completions/max_terminated_length": 54.6, + "completions/mean_length": 391.825, + "completions/mean_terminated_length": 54.6, + "completions/min_length": 334.6, + "completions/min_terminated_length": 54.6, + "entropy": 0.9223409309983254, + "epoch": 8.333333333333334, + "frac_reward_zero_std": 1.0, + "grad_norm": 0.0, + "learning_rate": 8e-08, + "loss": 0.0, + "num_tokens": 1602540.0, + "reward": 0.044499991834163664, + "reward_std": 0.8610172867774963, + "rewards/reward_fn/mean": 0.044499991834163664, + "rewards/reward_fn/std": 0.8610173106193543, + "step": 250, + "step_time": 33.55163867890042 + } + ], + "logging_steps": 10, + "max_steps": 250, + "num_input_tokens_seen": 1602540, + "num_train_epochs": 9, + "save_steps": 100, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": true + }, + "attributes": {} + } + }, + "total_flos": 0.0, + "train_batch_size": 2, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-250/training_args.bin b/checkpoint-250/training_args.bin new file mode 100644 index 0000000..bafd374 --- /dev/null +++ b/checkpoint-250/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0f0dd240a760a35edcf19bfedd721f31bea53099058c16d322d918d5f830b64a +size 7057 diff --git a/config.json b/config.json new file mode 100644 index 0000000..c920646 --- /dev/null +++ b/config.json @@ -0,0 +1,57 @@ +{ + "architectures": [ + "Qwen2ForCausalLM" + ], + "attention_dropout": 0.0, + "bos_token_id": null, + "dtype": "bfloat16", + "eos_token_id": 151645, + "hidden_act": "silu", + "hidden_size": 896, + "initializer_range": 0.02, + "intermediate_size": 4864, + "layer_types": [ + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention" + ], + "max_position_embeddings": 32768, + "max_window_layers": 21, + "model_type": "qwen2", + "num_attention_heads": 14, + "num_hidden_layers": 24, + "num_key_value_heads": 2, + "pad_token_id": 151643, + "rms_norm_eps": 1e-06, + "rope_parameters": { + "rope_theta": 1000000.0, + "rope_type": "default" + }, + "sliding_window": null, + "tie_word_embeddings": true, + "transformers_version": "5.0.0", + "use_cache": false, + "use_sliding_window": false, + "vocab_size": 151936 +} diff --git a/generation_config.json b/generation_config.json new file mode 100644 index 0000000..d58659f --- /dev/null +++ b/generation_config.json @@ -0,0 +1,13 @@ +{ + "do_sample": true, + "eos_token_id": [ + 151645, + 151643 + ], + "pad_token_id": 151643, + "repetition_penalty": 1.1, + "temperature": 0.7, + "top_k": 20, + "top_p": 0.8, + "transformers_version": "5.0.0" +} diff --git a/model.safetensors b/model.safetensors new file mode 100644 index 0000000..76221d2 --- /dev/null +++ b/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2f6ecf368a5199e23b82403ab68284fec38fb1f66520d261e9e8cb1f6b4ede31 +size 988097824 diff --git a/tokenizer.json b/tokenizer.json new file mode 100644 index 0000000..34510ff --- /dev/null +++ b/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3fd169731d2cbde95e10bf356d66d5997fd885dd8dbb6fb4684da3f23b2585d8 +size 11421892 diff --git a/tokenizer_config.json b/tokenizer_config.json new file mode 100644 index 0000000..7d75d3b --- /dev/null +++ b/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|im_end|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/training_args.bin b/training_args.bin new file mode 100644 index 0000000..bafd374 --- /dev/null +++ b/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0f0dd240a760a35edcf19bfedd721f31bea53099058c16d322d918d5f830b64a +size 7057