From 09a35f6b7b1f665594bb98fc842b646d69307db0 Mon Sep 17 00:00:00 2001 From: ModelHub XC Date: Sun, 6 Sep 2026 08:19:14 +0800 Subject: [PATCH] =?UTF-8?q?=E5=88=9D=E5=A7=8B=E5=8C=96=E9=A1=B9=E7=9B=AE?= =?UTF-8?q?=EF=BC=8C=E7=94=B1ModelHub=20XC=E7=A4=BE=E5=8C=BA=E6=8F=90?= =?UTF-8?q?=E4=BE=9B=E6=A8=A1=E5=9E=8B?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Model: aminediroHF/async-grpo-ckpt-smoke-r1d-1.5b Source: Original Platform --- .gitattributes | 36 ++++ README.md | 70 +++++++ chat_template.jinja | 1 + config.json | 62 ++++++ generation_config.json | 12 ++ last-checkpoint/chat_template.jinja | 1 + last-checkpoint/config.json | 62 ++++++ last-checkpoint/generation_config.json | 12 ++ last-checkpoint/model.safetensors | 3 + last-checkpoint/optimizer.pt | 3 + last-checkpoint/rng_state.pth | 3 + last-checkpoint/rollout_state.json | 1 + last-checkpoint/scheduler.pt | 3 + last-checkpoint/tokenizer.json | 3 + last-checkpoint/tokenizer_config.json | 26 +++ last-checkpoint/trainer_state.json | 268 +++++++++++++++++++++++++ last-checkpoint/training_args.bin | 3 + model.safetensors | 3 + tokenizer.json | 3 + tokenizer_config.json | 26 +++ training_args.bin | 3 + 21 files changed, 604 insertions(+) create mode 100644 .gitattributes create mode 100644 README.md create mode 100644 chat_template.jinja create mode 100644 config.json create mode 100644 generation_config.json create mode 100644 last-checkpoint/chat_template.jinja create mode 100644 last-checkpoint/config.json create mode 100644 last-checkpoint/generation_config.json create mode 100644 last-checkpoint/model.safetensors create mode 100644 last-checkpoint/optimizer.pt create mode 100644 last-checkpoint/rng_state.pth create mode 100644 last-checkpoint/rollout_state.json create mode 100644 last-checkpoint/scheduler.pt create mode 100644 last-checkpoint/tokenizer.json create mode 100644 last-checkpoint/tokenizer_config.json create mode 100644 last-checkpoint/trainer_state.json create mode 100644 last-checkpoint/training_args.bin create mode 100644 model.safetensors create mode 100644 tokenizer.json create mode 100644 tokenizer_config.json create mode 100644 training_args.bin diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..52373fe --- /dev/null +++ b/.gitattributes @@ -0,0 +1,36 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text +tokenizer.json filter=lfs diff=lfs merge=lfs -text diff --git a/README.md b/README.md new file mode 100644 index 0000000..d0398f8 --- /dev/null +++ b/README.md @@ -0,0 +1,70 @@ +--- +base_model: deepseek-ai/DeepSeek-R1-Distill-Qwen-1.5B +library_name: transformers +model_name: async-grpo-ckpt-smoke-r1d-1.5b +tags: +- generated_from_trainer +- trackio:https://huggingface.co/spaces/aminediroHF/async-grpo-ckpt-smoke-static-053862 +- trackio +- hf_jobs +- async-grpo +- trl +licence: license +--- + +# Model Card for async-grpo-ckpt-smoke-r1d-1.5b + +This model is a fine-tuned version of [deepseek-ai/DeepSeek-R1-Distill-Qwen-1.5B](https://huggingface.co/deepseek-ai/DeepSeek-R1-Distill-Qwen-1.5B). +It has been trained using [TRL](https://github.com/huggingface/trl). + +## Quick start + +```python +from transformers import pipeline + +question = "If you had a time machine, but could only go to the past or the future once and never return, which would you choose and why?" +generator = pipeline("text-generation", model="aminediroHF/async-grpo-ckpt-smoke-r1d-1.5b", device="cuda") +output = generator([{"role": "user", "content": question}], max_new_tokens=128, return_full_text=False)[0] +print(output["generated_text"]) +``` + +## Training procedure + + + + + +This model was trained with AsyncGRPO, a method introduced in [DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models](https://huggingface.co/papers/2402.03300). + +### Framework versions + +- TRL: 1.11.0.dev0 +- Transformers: 5.15.0 +- Pytorch: 2.13.0+cu130 +- Datasets: 5.0.1 +- Tokenizers: 0.22.2 + +## Citations + +Cite AsyncGRPO as: + +```bibtex +@article{shao2024deepseekmath, + title = {{DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models}}, + author = {Zhihong Shao and Peiyi Wang and Qihao Zhu and Runxin Xu and Junxiao Song and Mingchuan Zhang and Y. K. Li and Y. Wu and Daya Guo}, + year = 2024, + eprint = {arXiv:2402.03300}, +} +``` + +Cite TRL as: + +```bibtex +@software{vonwerra2020trl, + title = {{TRL: Transformers Reinforcement Learning}}, + author = {von Werra, Leandro and Belkada, Younes and Tunstall, Lewis and Beeching, Edward and Thrush, Tristan and Lambert, Nathan and Huang, Shengyi and Rasul, Kashif and Gallouédec, Quentin}, + license = {Apache-2.0}, + url = {https://github.com/huggingface/trl}, + year = {2020} +} +``` \ No newline at end of file diff --git a/chat_template.jinja b/chat_template.jinja new file mode 100644 index 0000000..c2066bd --- /dev/null +++ b/chat_template.jinja @@ -0,0 +1 @@ +{% if not add_generation_prompt is defined %}{% set add_generation_prompt = false %}{% endif %}{% set ns = namespace(is_first=false, is_tool=false, is_output_first=true, system_prompt='') %}{%- for message in messages %}{%- if message['role'] == 'system' %}{% set ns.system_prompt = message['content'] %}{%- endif %}{%- endfor %}{{bos_token}}{{ns.system_prompt}}{%- for message in messages %}{%- if message['role'] == 'user' %}{%- set ns.is_tool = false -%}{{'<|User|>' + message['content']}}{%- endif %}{%- if message['role'] == 'assistant' and message['content'] is none %}{%- set ns.is_tool = false -%}{%- for tool in message['tool_calls']%}{%- if not ns.is_first %}{{'<|Assistant|><|tool▁calls▁begin|><|tool▁call▁begin|>' + tool['type'] + '<|tool▁sep|>' + tool['function']['name'] + '\n' + '```json' + '\n' + tool['function']['arguments'] + '\n' + '```' + '<|tool▁call▁end|>'}}{%- set ns.is_first = true -%}{%- else %}{{'\n' + '<|tool▁call▁begin|>' + tool['type'] + '<|tool▁sep|>' + tool['function']['name'] + '\n' + '```json' + '\n' + tool['function']['arguments'] + '\n' + '```' + '<|tool▁call▁end|>'}}{{'<|tool▁calls▁end|><|end▁of▁sentence|>'}}{%- endif %}{%- endfor %}{%- endif %}{%- if message['role'] == 'assistant' and message['content'] is not none %}{%- if ns.is_tool %}{{'<|tool▁outputs▁end|>' + message['content'] + '<|end▁of▁sentence|>'}}{%- set ns.is_tool = false -%}{%- else %}{% set content = message['content'] %}{% if '' in content %}{% set content = content.split('')[-1] %}{% endif %}{{'<|Assistant|>' + content + '<|end▁of▁sentence|>'}}{%- endif %}{%- endif %}{%- if message['role'] == 'tool' %}{%- set ns.is_tool = true -%}{%- if ns.is_output_first %}{{'<|tool▁outputs▁begin|><|tool▁output▁begin|>' + message['content'] + '<|tool▁output▁end|>'}}{%- set ns.is_output_first = false %}{%- else %}{{'\n<|tool▁output▁begin|>' + message['content'] + '<|tool▁output▁end|>'}}{%- endif %}{%- endif %}{%- endfor -%}{% if ns.is_tool %}{{'<|tool▁outputs▁end|>'}}{% endif %}{% if add_generation_prompt and not ns.is_tool %}{{'<|Assistant|>\n'}}{% endif %} \ No newline at end of file diff --git a/config.json b/config.json new file mode 100644 index 0000000..4e5bff4 --- /dev/null +++ b/config.json @@ -0,0 +1,62 @@ +{ + "architectures": [ + "Qwen2ForCausalLM" + ], + "attention_dropout": 0.0, + "bos_token_id": 151646, + "dtype": "float32", + "eos_token_id": 151643, + "hidden_act": "silu", + "hidden_size": 1536, + "initializer_range": 0.02, + "intermediate_size": 8960, + "layer_types": [ + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention" + ], + "max_position_embeddings": 131072, + "max_window_layers": 21, + "model_type": "qwen2", + "num_attention_heads": 12, + "num_hidden_layers": 28, + "num_key_value_heads": 2, + "pad_token_id": 151643, + "rms_norm_eps": 1e-06, + "rope_parameters": { + "rope_theta": 10000, + "rope_type": "default" + }, + "sliding_window": null, + "tie_word_embeddings": false, + "transformers_version": "5.15.0", + "use_cache": false, + "use_mrope": false, + "use_sliding_window": false, + "vocab_size": 151936 +} diff --git a/generation_config.json b/generation_config.json new file mode 100644 index 0000000..d75b285 --- /dev/null +++ b/generation_config.json @@ -0,0 +1,12 @@ +{ + "_from_model_config": true, + "bos_token_id": 151646, + "do_sample": true, + "eos_token_id": [ + 151643 + ], + "pad_token_id": 151643, + "temperature": 0.6, + "top_p": 0.95, + "transformers_version": "5.15.0" +} diff --git a/last-checkpoint/chat_template.jinja b/last-checkpoint/chat_template.jinja new file mode 100644 index 0000000..c2066bd --- /dev/null +++ b/last-checkpoint/chat_template.jinja @@ -0,0 +1 @@ +{% if not add_generation_prompt is defined %}{% set add_generation_prompt = false %}{% endif %}{% set ns = namespace(is_first=false, is_tool=false, is_output_first=true, system_prompt='') %}{%- for message in messages %}{%- if message['role'] == 'system' %}{% set ns.system_prompt = message['content'] %}{%- endif %}{%- endfor %}{{bos_token}}{{ns.system_prompt}}{%- for message in messages %}{%- if message['role'] == 'user' %}{%- set ns.is_tool = false -%}{{'<|User|>' + message['content']}}{%- endif %}{%- if message['role'] == 'assistant' and message['content'] is none %}{%- set ns.is_tool = false -%}{%- for tool in message['tool_calls']%}{%- if not ns.is_first %}{{'<|Assistant|><|tool▁calls▁begin|><|tool▁call▁begin|>' + tool['type'] + '<|tool▁sep|>' + tool['function']['name'] + '\n' + '```json' + '\n' + tool['function']['arguments'] + '\n' + '```' + '<|tool▁call▁end|>'}}{%- set ns.is_first = true -%}{%- else %}{{'\n' + '<|tool▁call▁begin|>' + tool['type'] + '<|tool▁sep|>' + tool['function']['name'] + '\n' + '```json' + '\n' + tool['function']['arguments'] + '\n' + '```' + '<|tool▁call▁end|>'}}{{'<|tool▁calls▁end|><|end▁of▁sentence|>'}}{%- endif %}{%- endfor %}{%- endif %}{%- if message['role'] == 'assistant' and message['content'] is not none %}{%- if ns.is_tool %}{{'<|tool▁outputs▁end|>' + message['content'] + '<|end▁of▁sentence|>'}}{%- set ns.is_tool = false -%}{%- else %}{% set content = message['content'] %}{% if '' in content %}{% set content = content.split('')[-1] %}{% endif %}{{'<|Assistant|>' + content + '<|end▁of▁sentence|>'}}{%- endif %}{%- endif %}{%- if message['role'] == 'tool' %}{%- set ns.is_tool = true -%}{%- if ns.is_output_first %}{{'<|tool▁outputs▁begin|><|tool▁output▁begin|>' + message['content'] + '<|tool▁output▁end|>'}}{%- set ns.is_output_first = false %}{%- else %}{{'\n<|tool▁output▁begin|>' + message['content'] + '<|tool▁output▁end|>'}}{%- endif %}{%- endif %}{%- endfor -%}{% if ns.is_tool %}{{'<|tool▁outputs▁end|>'}}{% endif %}{% if add_generation_prompt and not ns.is_tool %}{{'<|Assistant|>\n'}}{% endif %} \ No newline at end of file diff --git a/last-checkpoint/config.json b/last-checkpoint/config.json new file mode 100644 index 0000000..4e5bff4 --- /dev/null +++ b/last-checkpoint/config.json @@ -0,0 +1,62 @@ +{ + "architectures": [ + "Qwen2ForCausalLM" + ], + "attention_dropout": 0.0, + "bos_token_id": 151646, + "dtype": "float32", + "eos_token_id": 151643, + "hidden_act": "silu", + "hidden_size": 1536, + "initializer_range": 0.02, + "intermediate_size": 8960, + "layer_types": [ + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention" + ], + "max_position_embeddings": 131072, + "max_window_layers": 21, + "model_type": "qwen2", + "num_attention_heads": 12, + "num_hidden_layers": 28, + "num_key_value_heads": 2, + "pad_token_id": 151643, + "rms_norm_eps": 1e-06, + "rope_parameters": { + "rope_theta": 10000, + "rope_type": "default" + }, + "sliding_window": null, + "tie_word_embeddings": false, + "transformers_version": "5.15.0", + "use_cache": false, + "use_mrope": false, + "use_sliding_window": false, + "vocab_size": 151936 +} diff --git a/last-checkpoint/generation_config.json b/last-checkpoint/generation_config.json new file mode 100644 index 0000000..d75b285 --- /dev/null +++ b/last-checkpoint/generation_config.json @@ -0,0 +1,12 @@ +{ + "_from_model_config": true, + "bos_token_id": 151646, + "do_sample": true, + "eos_token_id": [ + 151643 + ], + "pad_token_id": 151643, + "temperature": 0.6, + "top_p": 0.95, + "transformers_version": "5.15.0" +} diff --git a/last-checkpoint/model.safetensors b/last-checkpoint/model.safetensors new file mode 100644 index 0000000..a7bdd2e --- /dev/null +++ b/last-checkpoint/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b9c47b492488c981054bd27ecc224d98bf429a7b0e0c1f09826d988b43ae4a8b +size 7108390424 diff --git a/last-checkpoint/optimizer.pt b/last-checkpoint/optimizer.pt new file mode 100644 index 0000000..98a085c --- /dev/null +++ b/last-checkpoint/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ed8a5af882a4788a95cebc2d8051545a8aa7c63d14fad89365e01f5a1e623db8 +size 14217004240 diff --git a/last-checkpoint/rng_state.pth b/last-checkpoint/rng_state.pth new file mode 100644 index 0000000..1feba1a --- /dev/null +++ b/last-checkpoint/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:098b29492211804ab324a36f37466821d948280bb74fce4ba895c03f13ecd878 +size 14645 diff --git a/last-checkpoint/rollout_state.json b/last-checkpoint/rollout_state.json new file mode 100644 index 0000000..8b7d2c0 --- /dev/null +++ b/last-checkpoint/rollout_state.json @@ -0,0 +1 @@ +{"rows_consumed": 56, "groups_trained": 18} \ No newline at end of file diff --git a/last-checkpoint/scheduler.pt b/last-checkpoint/scheduler.pt new file mode 100644 index 0000000..2de350e --- /dev/null +++ b/last-checkpoint/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8ab327a1048744a1568a3c4ad6453c9b156bc3dcfcd06c5a301e6c619985da55 +size 1465 diff --git a/last-checkpoint/tokenizer.json b/last-checkpoint/tokenizer.json new file mode 100644 index 0000000..4306d79 --- /dev/null +++ b/last-checkpoint/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:322664cdc3082b6eba003af5228a77ca1d7936d402e584ecde8f15d3d98bdb72 +size 11421911 diff --git a/last-checkpoint/tokenizer_config.json b/last-checkpoint/tokenizer_config.json new file mode 100644 index 0000000..7b30a3a --- /dev/null +++ b/last-checkpoint/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "bos_token": "<|begin▁of▁sentence|>", + "clean_up_tokenization_spaces": false, + "eos_token": "<|end▁of▁sentence|>", + "is_local": false, + "legacy": true, + "local_files_only": false, + "model_max_length": 16384, + "pad_token": "<|end▁of▁sentence|>", + "response_template": { + "defaults": { + "role": "assistant" + }, + "fields": { + "content": { + "close_pattern": "<|end▁of▁sentence|>\\s*", + "content": "text" + } + }, + "start_anchor": "<|Assistant|>\n" + }, + "sp_model_kwargs": {}, + "tokenizer_class": "TokenizersBackend", + "unk_token": null +} diff --git a/last-checkpoint/trainer_state.json b/last-checkpoint/trainer_state.json new file mode 100644 index 0000000..2498b41 --- /dev/null +++ b/last-checkpoint/trainer_state.json @@ -0,0 +1,268 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.5, + "eval_steps": 500, + "global_step": 8, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "buffer_qsize": 0.0, + "clip_ratio/high_max": 3.222041777917184e-05, + "clip_ratio/high_mean": 3.222041777917184e-05, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 3.222041777917184e-05, + "completions/mean_length": 3876.25, + "entropy": 0.9389231279492378, + "epoch": 0.25, + "forward_time_s": 0.30089330673217773, + "generation_tok_per_s": 6915.577392578125, + "grad_norm": 0.36549293994903564, + "kl": 0.0003301510150777176, + "learning_rate": 1e-06, + "loss": 0.02711719088256359, + "queue_wait_time_s": 2.0328194051980972, + "ratio": 1.0000957623124123, + "reward": 0.4375, + "reward_std": 0.49206146597862244, + "rewards/r1_distill_math_reward": 0.4375, + "scoring_time_ms": 18.674388885498047, + "step": 1, + "step_time": 6.262407066999003, + "train_seq_len": 4371.125, + "training_tok/s": 12972.816542826044, + "wait_scoring_ms": 0.20329749584197998, + "weight_sync_time_s": 0.14123249053955078 + }, + { + "buffer_qsize": 3.5, + "clip_ratio/high_max": 1.6543144738534465e-05, + "clip_ratio/high_mean": 1.6543144738534465e-05, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 1.6543144738534465e-05, + "completions/mean_length": 6120.375, + "entropy": 0.7296525910496712, + "epoch": 0.5, + "forward_time_s": 0.25667285919189453, + "generation_tok_per_s": 13383.16943359375, + "grad_norm": 0.2867599427700043, + "iteration_time_s": 7.6814478730084375, + "kl": 0.00025582036323612556, + "learning_rate": 1e-06, + "loss": 3.546476364135742e-05, + "queue_wait_time_s": 0.3530673235654831, + "ratio": 1.0000218078494072, + "reward": 0.75, + "reward_std": 0.4330126941204071, + "rewards/r1_distill_math_reward": 0.75, + "scoring_time_ms": 5.158496618270874, + "step": 2, + "step_time": 7.5191129069717135, + "train_seq_len": 6374.125, + "training_tok/s": 13318.513123161385, + "wait_scoring_ms": 5.808052182197571 + }, + { + "buffer_qsize": 18.0, + "clip_ratio/high_max": 7.113443098205607e-05, + "clip_ratio/high_mean": 7.113443098205607e-05, + "clip_ratio/low_mean": 1.5646326573914848e-05, + "clip_ratio/low_min": 1.5646326573914848e-05, + "clip_ratio/region_mean": 8.678075755597092e-05, + "completions/mean_length": 6053.8125, + "entropy": 0.8014494031667709, + "epoch": 0.75, + "forward_time_s": 0.2557235360145569, + "generation_tok_per_s": 13382.43359375, + "grad_norm": 0.3024653494358063, + "iteration_time_s": 7.571961241017561, + "kl": 0.00028099300106987357, + "learning_rate": 1e-06, + "loss": 0.00955304503440857, + "queue_wait_time_s": 0.00046597421169281006, + "ratio": 1.0000539049506187, + "reward": 0.5625, + "reward_std": 0.38186579942703247, + "rewards/r1_distill_math_reward": 0.5625, + "scoring_time_ms": 0.7072950005531311, + "step": 3, + "step_time": 7.4889817090006545, + "train_seq_len": 6390.375, + "training_tok/s": 13393.731171849211, + "wait_scoring_ms": 12.160233974456787 + }, + { + "buffer_qsize": 34.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 4.13285370086669e-05, + "clip_ratio/low_min": 4.13285370086669e-05, + "clip_ratio/region_mean": 4.13285370086669e-05, + "completions/mean_length": 6813.625, + "entropy": 0.811307743191719, + "epoch": 1.0, + "forward_time_s": 0.2839837670326233, + "generation_tok_per_s": 13381.1396484375, + "grad_norm": 0.2735491394996643, + "iteration_time_s": 8.428669046988944, + "kl": 0.00027251767460256815, + "learning_rate": 1e-06, + "loss": 0.038605645298957825, + "queue_wait_time_s": 0.0004654228687286377, + "ratio": 0.9999778047204018, + "reward": 0.25, + "reward_std": 0.4330126941204071, + "rewards/r1_distill_math_reward": 0.25, + "scoring_time_ms": 1.8631829619407654, + "step": 4, + "step_time": 8.344294181006262, + "train_seq_len": 7421.5, + "training_tok/s": 13408.494449990198, + "wait_scoring_ms": 14.634688377380371, + "weight_sync_time_s": 0.15913748741149902 + }, + { + "buffer_qsize": 0.0, + "clip_ratio/high_max": 1.8846135390049312e-05, + "clip_ratio/high_mean": 1.8846135390049312e-05, + "clip_ratio/low_mean": 1.011736094369553e-05, + "clip_ratio/low_min": 1.011736094369553e-05, + "clip_ratio/region_mean": 2.8963496333744843e-05, + "completions/mean_length": 5484.0, + "entropy": 0.7321677580475807, + "epoch": 0.125, + "forward_time_s": 0.3462470471858978, + "generation_tok_per_s": 9634.451171875, + "grad_norm": 0.28255149722099304, + "kl": 0.00025452028421568684, + "learning_rate": 1e-06, + "loss": -0.004137326031923294, + "queue_wait_time_s": 1.8160016685724258, + "ratio": 1.0000781491398811, + "reward": 0.5625, + "reward_std": 0.38186579942703247, + "rewards/r1_distill_math_reward": 0.5625, + "scoring_time_ms": 13.104030728340149, + "step": 5, + "step_time": 7.895209455047734, + "train_seq_len": 5787.375, + "training_tok/s": 13242.969307036337, + "wait_scoring_ms": 0.1778009943664074, + "weight_sync_time_s": 0.13536810874938965 + }, + { + "buffer_qsize": 3.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 0.0, + "completions/mean_length": 6134.3125, + "entropy": 0.6220929026603699, + "epoch": 0.25, + "forward_time_s": 0.26168057322502136, + "generation_tok_per_s": 11878.65380859375, + "grad_norm": 0.17205984890460968, + "iteration_time_s": 7.687154909974197, + "kl": 0.00023902823340904433, + "learning_rate": 1e-06, + "loss": 0.0026915371417999268, + "queue_wait_time_s": 0.3177753835916519, + "ratio": 0.9999898672103882, + "reward": 0.375, + "reward_std": 0.21650634706020355, + "rewards/r1_distill_math_reward": 0.375, + "scoring_time_ms": 8.408703327178955, + "step": 6, + "step_time": 7.5453005689778365, + "train_seq_len": 6619.25, + "training_tok/s": 13203.215699232549, + "wait_scoring_ms": 0.5662760138511658 + }, + { + "buffer_qsize": 18.0, + "clip_ratio/high_max": 5.6621894145791885e-05, + "clip_ratio/high_mean": 5.6621894145791885e-05, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/region_mean": 5.6621894145791885e-05, + "completions/mean_length": 5713.375, + "entropy": 0.8666651099920273, + "epoch": 0.375, + "forward_time_s": 0.2426280975341797, + "generation_tok_per_s": 13567.3173828125, + "grad_norm": 0.30514439940452576, + "iteration_time_s": 7.157755050022388, + "kl": 0.0002972106722154422, + "learning_rate": 1e-06, + "loss": 0.015170708298683167, + "queue_wait_time_s": 0.0004561096429824829, + "ratio": 1.0000762566924095, + "reward": 0.75, + "reward_std": 0.40742091834545135, + "rewards/r1_distill_math_reward": 0.75, + "scoring_time_ms": 2.2001885175704956, + "step": 7, + "step_time": 7.077003714046441, + "train_seq_len": 5968.5, + "training_tok/s": 13234.408437249378, + "wait_scoring_ms": 14.30916166305542 + }, + { + "buffer_qsize": 34.0, + "clip_ratio/high_max": 9.804690307646524e-06, + "clip_ratio/high_mean": 9.804690307646524e-06, + "clip_ratio/low_mean": 3.2692873901396524e-05, + "clip_ratio/low_min": 3.2692873901396524e-05, + "clip_ratio/region_mean": 4.249756420904305e-05, + "completions/mean_length": 5408.0625, + "entropy": 0.9108624383807182, + "epoch": 0.5, + "forward_time_s": 0.23351335525512695, + "generation_tok_per_s": 13553.60693359375, + "grad_norm": 0.3738342225551605, + "iteration_time_s": 6.897600525000598, + "kl": 0.0003173556542606093, + "learning_rate": 1e-06, + "loss": 0.0034758001565933228, + "queue_wait_time_s": 0.0004056990146636963, + "ratio": 0.9998656511306763, + "reward": 0.5, + "reward_std": 0.4330126941204071, + "rewards/r1_distill_math_reward": 0.5, + "scoring_time_ms": 18.20103907585144, + "step": 8, + "step_time": 6.822746234043734, + "train_seq_len": 5848.75, + "training_tok/s": 13207.699007613326, + "wait_scoring_ms": 35.41268348693848, + "weight_sync_time_s": 0.23273587226867676 + } + ], + "logging_steps": 1, + "max_steps": 8, + "num_input_tokens_seen": 0, + "num_train_epochs": 9223372036854775807, + "save_steps": 4, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": true + }, + "attributes": {} + } + }, + "total_flos": 6909415142986752.0, + "train_batch_size": 2, + "trial_name": null, + "trial_params": null +} diff --git a/last-checkpoint/training_args.bin b/last-checkpoint/training_args.bin new file mode 100644 index 0000000..977e7be --- /dev/null +++ b/last-checkpoint/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5ab9aeab1b506ada6e8f21d2b4d576c5b644e131107ebbd7055c5587655acf4b +size 6033 diff --git a/model.safetensors b/model.safetensors new file mode 100644 index 0000000..912a4f8 --- /dev/null +++ b/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:52f925bd55fdae95b0027ece929c8659de0e81f79d22d76ff9535ba7a2acda90 +size 7108390424 diff --git a/tokenizer.json b/tokenizer.json new file mode 100644 index 0000000..4306d79 --- /dev/null +++ b/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:322664cdc3082b6eba003af5228a77ca1d7936d402e584ecde8f15d3d98bdb72 +size 11421911 diff --git a/tokenizer_config.json b/tokenizer_config.json new file mode 100644 index 0000000..7b30a3a --- /dev/null +++ b/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "bos_token": "<|begin▁of▁sentence|>", + "clean_up_tokenization_spaces": false, + "eos_token": "<|end▁of▁sentence|>", + "is_local": false, + "legacy": true, + "local_files_only": false, + "model_max_length": 16384, + "pad_token": "<|end▁of▁sentence|>", + "response_template": { + "defaults": { + "role": "assistant" + }, + "fields": { + "content": { + "close_pattern": "<|end▁of▁sentence|>\\s*", + "content": "text" + } + }, + "start_anchor": "<|Assistant|>\n" + }, + "sp_model_kwargs": {}, + "tokenizer_class": "TokenizersBackend", + "unk_token": null +} diff --git a/training_args.bin b/training_args.bin new file mode 100644 index 0000000..863eb1f --- /dev/null +++ b/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d7dae1dc2eb45fb4b5a0a33e34e6ad39d4eab678a33f05e303250ca5e217bfbe +size 6033