{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 0.8695652173913043, "eval_steps": 500, "global_step": 10, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 84.5, "completions/max_terminated_length": 84.5, "completions/mean_length": 46.125, "completions/mean_terminated_length": 46.125, "completions/min_length": 20.5, "completions/min_terminated_length": 20.5, "epoch": 0.17391304347826086, "frac_reward_zero_std": 0.0, "grad_norm": 50.021663665771484, "kl": 0.0, "learning_rate": 5e-07, "loss": -0.0009, "num_tokens": 1393.0, "reward": 0.3569868355989456, "reward_std": 0.10140731558203697, "rewards/concensus_correctness_reward_func/mean": 0.0, "rewards/concensus_correctness_reward_func/std": 0.0, "rewards/consensus_reward_func/mean": 0.0, "rewards/consensus_reward_func/std": 0.0, "rewards/cumulative_reward_2/mean": 0.0, "rewards/cumulative_reward_2/std": 0.0, "rewards/final_correctness_reward_func/mean": 0.0, "rewards/final_correctness_reward_func/std": 0.0, "rewards/question_recreation_reward_func/mean": 0.032236830331385136, "rewards/question_recreation_reward_func/std": 0.017948506865650415, "rewards/soft_format_reward_func/mean": 0.0, "rewards/soft_format_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.0, "rewards/strict_format_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.32474999129772186, "rewards/xmlcount_reward_func/std": 0.19420991092920303, "step": 2 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 109.0, "completions/max_terminated_length": 109.0, "completions/mean_length": 70.75, "completions/mean_terminated_length": 70.75, "completions/min_length": 41.5, "completions/min_terminated_length": 41.5, "epoch": 0.34782608695652173, "frac_reward_zero_std": 0.0, "grad_norm": 19.0537109375, "kl": 0.002165338257327676, "learning_rate": 4.415111107797445e-07, "loss": 0.0282, "num_tokens": 2983.0, "reward": 0.4277445673942566, "reward_std": 0.14516344666481018, "rewards/concensus_correctness_reward_func/mean": 0.0, "rewards/concensus_correctness_reward_func/std": 0.0, "rewards/consensus_reward_func/mean": 0.0, "rewards/consensus_reward_func/std": 0.0, "rewards/cumulative_reward_2/mean": 0.0, "rewards/cumulative_reward_2/std": 0.0, "rewards/final_correctness_reward_func/mean": 0.0, "rewards/final_correctness_reward_func/std": 0.0, "rewards/question_recreation_reward_func/mean": 0.16299461387097836, "rewards/question_recreation_reward_func/std": 0.06881882064044476, "rewards/soft_format_reward_func/mean": 0.0, "rewards/soft_format_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.0, "rewards/strict_format_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.26474998891353607, "rewards/xmlcount_reward_func/std": 0.28970315866172314, "step": 4 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 177.0, "completions/max_terminated_length": 177.0, "completions/mean_length": 91.625, "completions/mean_terminated_length": 91.625, "completions/min_length": 33.5, "completions/min_terminated_length": 33.5, "epoch": 0.5217391304347826, "frac_reward_zero_std": 0.0, "grad_norm": 13.790047645568848, "kl": 0.0031155047909123823, "learning_rate": 2.934120444167326e-07, "loss": 0.1559, "num_tokens": 4740.0, "reward": 0.25488629564642906, "reward_std": 0.1210553664714098, "rewards/concensus_correctness_reward_func/mean": 0.0, "rewards/concensus_correctness_reward_func/std": 0.0, "rewards/consensus_reward_func/mean": 0.0, "rewards/consensus_reward_func/std": 0.0, "rewards/cumulative_reward_2/mean": 0.0, "rewards/cumulative_reward_2/std": 0.0, "rewards/final_correctness_reward_func/mean": 0.0, "rewards/final_correctness_reward_func/std": 0.0, "rewards/question_recreation_reward_func/mean": 0.04713628068566322, "rewards/question_recreation_reward_func/std": 0.032215227372944355, "rewards/soft_format_reward_func/mean": 0.0, "rewards/soft_format_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.0, "rewards/strict_format_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.20775000005960464, "rewards/xmlcount_reward_func/std": 0.2513107433915138, "step": 6 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 198.0, "completions/max_terminated_length": 98.5, "completions/mean_length": 95.5, "completions/mean_terminated_length": 69.5, "completions/min_length": 44.5, "completions/min_terminated_length": 44.5, "epoch": 0.6956521739130435, "frac_reward_zero_std": 0.0, "grad_norm": 19.808725357055664, "kl": 0.015141080017201602, "learning_rate": 1.2500000000000005e-07, "loss": 0.0002, "num_tokens": 6528.0, "reward": 0.2949913889169693, "reward_std": 0.13783530704677105, "rewards/concensus_correctness_reward_func/mean": 0.0, "rewards/concensus_correctness_reward_func/std": 0.0, "rewards/consensus_reward_func/mean": 0.0, "rewards/consensus_reward_func/std": 0.0, "rewards/cumulative_reward_2/mean": 0.0, "rewards/cumulative_reward_2/std": 0.0, "rewards/final_correctness_reward_func/mean": 0.0, "rewards/final_correctness_reward_func/std": 0.0, "rewards/question_recreation_reward_func/mean": 0.04011638090014458, "rewards/question_recreation_reward_func/std": 0.03736674599349499, "rewards/soft_format_reward_func/mean": 0.0, "rewards/soft_format_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.0, "rewards/strict_format_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.2548750042915344, "rewards/xmlcount_reward_func/std": 0.30075830966234207, "step": 8 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 63.5, "completions/max_terminated_length": 63.5, "completions/mean_length": 49.5, "completions/mean_terminated_length": 49.5, "completions/min_length": 24.5, "completions/min_terminated_length": 24.5, "epoch": 0.8695652173913043, "frac_reward_zero_std": 0.0, "grad_norm": 43.93717956542969, "kl": 0.027774153277277946, "learning_rate": 1.507684480352292e-08, "loss": -0.0862, "num_tokens": 7948.0, "reward": 0.4353444501757622, "reward_std": 0.1529897004365921, "rewards/concensus_correctness_reward_func/mean": 0.0, "rewards/concensus_correctness_reward_func/std": 0.0, "rewards/consensus_reward_func/mean": 0.0, "rewards/consensus_reward_func/std": 0.0, "rewards/cumulative_reward_2/mean": 0.0, "rewards/cumulative_reward_2/std": 0.0, "rewards/final_correctness_reward_func/mean": 0.0, "rewards/final_correctness_reward_func/std": 0.0, "rewards/question_recreation_reward_func/mean": 0.132094481959939, "rewards/question_recreation_reward_func/std": 0.03221887769177556, "rewards/soft_format_reward_func/mean": 0.0, "rewards/soft_format_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.0, "rewards/strict_format_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.3032499924302101, "rewards/xmlcount_reward_func/std": 0.16571738570928574, "step": 10 }, { "epoch": 0.8695652173913043, "step": 10, "total_flos": 0.0, "train_loss": 0.019416041672229767, "train_runtime": 1276.5475, "train_samples_per_second": 0.031, "train_steps_per_second": 0.008 } ], "logging_steps": 2, "max_steps": 10, "num_input_tokens_seen": 7948, "num_train_epochs": 1, "save_steps": 10, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 0.0, "train_batch_size": 2, "trial_name": null, "trial_params": null }