{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 0.05, "eval_steps": 500, "global_step": 500, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.14998279511928558, "epoch": 0.0001, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 2568.0, "reward": 0.0, "reward_std": 0.0, "rewards/_reward/mean": 0.0, "rewards/_reward/std": 0.0, "step": 1, "step_time": 12.454096379224211 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.3256736550894048, "epoch": 0.001, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 9.819999999999999e-07, "loss": 0.0, "num_tokens": 27464.0, "reward": 0.0, "reward_std": 0.0, "rewards/_reward/mean": 0.0, "rewards/_reward/std": 0.0, "step": 10, "step_time": 11.780018932263678 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.2880026239901781, "epoch": 0.002, "frac_reward_zero_std": 0.8, "grad_norm": 0.0, "learning_rate": 9.619999999999999e-07, "loss": 0.0, "num_tokens": 55576.0, "reward": 0.1, "reward_std": 0.09258201122283935, "rewards/_reward/mean": 0.1, "rewards/_reward/std": 0.09258201122283935, "step": 20, "step_time": 12.011081893160007 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9875, "completions/max_length": 256.0, "completions/max_terminated_length": 17.5, "completions/mean_length": 254.9875, "completions/mean_terminated_length": 17.5, "completions/min_length": 247.9, "completions/min_terminated_length": 17.5, "entropy": 0.28849115688353777, "epoch": 0.003, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 9.419999999999999e-07, "loss": 0.0, "num_tokens": 83367.0, "reward": 0.0, "reward_std": 0.0, "rewards/_reward/mean": 0.0, "rewards/_reward/std": 0.0, "step": 30, "step_time": 12.034194040577859 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.95, "completions/max_length": 256.0, "completions/max_terminated_length": 23.4, "completions/mean_length": 253.9, "completions/mean_terminated_length": 21.4, "completions/min_length": 249.1, "completions/min_terminated_length": 18.7, "entropy": 0.2561770185828209, "epoch": 0.004, "frac_reward_zero_std": 0.9, "grad_norm": 2.998258590698242, "learning_rate": 9.22e-07, "loss": 1.4901161193847657e-09, "num_tokens": 111439.0, "reward": 0.0875, "reward_std": 0.03535533845424652, "rewards/_reward/mean": 0.0875, "rewards/_reward/std": 0.03535533845424652, "step": 40, "step_time": 12.092716509709135 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.21435276679694654, "epoch": 0.005, "frac_reward_zero_std": 0.9, "grad_norm": 0.0, "learning_rate": 9.02e-07, "loss": 1.4901161193847657e-09, "num_tokens": 139943.0, "reward": 0.0375, "reward_std": 0.051754921674728394, "rewards/_reward/mean": 0.0375, "rewards/_reward/std": 0.051754921674728394, "step": 50, "step_time": 11.898367898073047 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.20879782736301422, "epoch": 0.006, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 8.82e-07, "loss": 0.0, "num_tokens": 168911.0, "reward": 0.0, "reward_std": 0.0, "rewards/_reward/mean": 0.0, "rewards/_reward/std": 0.0, "step": 60, "step_time": 11.92987802445423 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.23100027367472648, "epoch": 0.007, "frac_reward_zero_std": 0.9, "grad_norm": 0.0, "learning_rate": 8.62e-07, "loss": -1.4901161193847657e-09, "num_tokens": 197359.0, "reward": 0.0125, "reward_std": 0.03535533845424652, "rewards/_reward/mean": 0.0125, "rewards/_reward/std": 0.03535533845424652, "step": 70, "step_time": 11.958151014475153 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.24868577010929585, "epoch": 0.008, "frac_reward_zero_std": 0.8, "grad_norm": 2.808773994445801, "learning_rate": 8.419999999999999e-07, "loss": 0.0, "num_tokens": 224775.0, "reward": 0.15, "reward_std": 0.08711026012897491, "rewards/_reward/mean": 0.15, "rewards/_reward/std": 0.08711026012897491, "step": 80, "step_time": 11.96262693060562 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.23981668353080748, "epoch": 0.009, "frac_reward_zero_std": 0.9, "grad_norm": 0.0, "learning_rate": 8.219999999999999e-07, "loss": -1.4901161193847657e-09, "num_tokens": 252687.0, "reward": 0.0625, "reward_std": 0.051754921674728394, "rewards/_reward/mean": 0.0625, "rewards/_reward/std": 0.051754921674728394, "step": 90, "step_time": 11.79044756719377 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9875, "completions/max_length": 256.0, "completions/max_terminated_length": 25.6, "completions/mean_length": 256.0, "completions/mean_terminated_length": 25.6, "completions/min_length": 256.0, "completions/min_terminated_length": 25.6, "entropy": 0.2731012573465705, "epoch": 0.01, "frac_reward_zero_std": 0.9, "grad_norm": 0.0, "learning_rate": 8.02e-07, "loss": 0.0, "num_tokens": 280847.0, "reward": 0.0625, "reward_std": 0.051754921674728394, "rewards/_reward/mean": 0.0625, "rewards/_reward/std": 0.051754921674728394, "step": 100, "step_time": 11.873896015947684 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.2539512518793344, "epoch": 0.011, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 7.82e-07, "loss": 0.0, "num_tokens": 308583.0, "reward": 0.0, "reward_std": 0.0, "rewards/_reward/mean": 0.0, "rewards/_reward/std": 0.0, "step": 110, "step_time": 12.493365852814168 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.23124769441783427, "epoch": 0.012, "frac_reward_zero_std": 0.8, "grad_norm": 0.0, "learning_rate": 7.62e-07, "loss": -4.470348358154297e-09, "num_tokens": 336751.0, "reward": 0.0375, "reward_std": 0.0816463440656662, "rewards/_reward/mean": 0.0375, "rewards/_reward/std": 0.0816463440656662, "step": 120, "step_time": 12.216748453862966 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.24605347998440266, "epoch": 0.013, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 7.42e-07, "loss": 0.0, "num_tokens": 366007.0, "reward": 0.0, "reward_std": 0.0, "rewards/_reward/mean": 0.0, "rewards/_reward/std": 0.0, "step": 130, "step_time": 11.979620195319876 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.23477237336337567, "epoch": 0.014, "frac_reward_zero_std": 0.9, "grad_norm": 0.0, "learning_rate": 7.219999999999999e-07, "loss": -1.4901161193847657e-09, "num_tokens": 396623.0, "reward": 0.0125, "reward_std": 0.03535533845424652, "rewards/_reward/mean": 0.0125, "rewards/_reward/std": 0.03535533845424652, "step": 140, "step_time": 11.984532529045827 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.19286549724638463, "epoch": 0.015, "frac_reward_zero_std": 0.7, "grad_norm": 0.0, "learning_rate": 7.019999999999999e-07, "loss": 1.4901161193847657e-09, "num_tokens": 425055.0, "reward": 0.0875, "reward_std": 0.1388651818037033, "rewards/_reward/mean": 0.0875, "rewards/_reward/std": 0.1388651818037033, "step": 150, "step_time": 11.88744165727403 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.2272243991494179, "epoch": 0.016, "frac_reward_zero_std": 0.7, "grad_norm": 3.0967137813568115, "learning_rate": 6.82e-07, "loss": 0.0, "num_tokens": 453151.0, "reward": 0.15, "reward_std": 0.1334012657403946, "rewards/_reward/mean": 0.15, "rewards/_reward/std": 0.1334012657403946, "step": 160, "step_time": 12.118341535096988 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.19670370109379293, "epoch": 0.017, "frac_reward_zero_std": 0.8, "grad_norm": 0.0, "learning_rate": 6.62e-07, "loss": 0.0, "num_tokens": 480895.0, "reward": 0.05, "reward_std": 0.08711026012897491, "rewards/_reward/mean": 0.05, "rewards/_reward/std": 0.08711026012897491, "step": 170, "step_time": 11.987511804816313 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.28401327803730964, "epoch": 0.018, "frac_reward_zero_std": 0.9, "grad_norm": 0.0, "learning_rate": 6.42e-07, "loss": -1.4901161193847657e-09, "num_tokens": 508823.0, "reward": 0.0625, "reward_std": 0.051754921674728394, "rewards/_reward/mean": 0.0625, "rewards/_reward/std": 0.051754921674728394, "step": 180, "step_time": 11.97194695000071 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.2276454884558916, "epoch": 0.019, "frac_reward_zero_std": 0.6, "grad_norm": 0.0, "learning_rate": 6.219999999999999e-07, "loss": -5.960464477539063e-09, "num_tokens": 537039.0, "reward": 0.075, "reward_std": 0.1632926881313324, "rewards/_reward/mean": 0.075, "rewards/_reward/std": 0.1632926881313324, "step": 190, "step_time": 11.945305320294574 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.21875664182007312, "epoch": 0.02, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 6.019999999999999e-07, "loss": 0.0, "num_tokens": 565071.0, "reward": 0.0, "reward_std": 0.0, "rewards/_reward/mean": 0.0, "rewards/_reward/std": 0.0, "step": 200, "step_time": 12.232728651817888 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9875, "completions/max_length": 256.0, "completions/max_terminated_length": 21.1, "completions/mean_length": 255.4375, "completions/mean_terminated_length": 21.1, "completions/min_length": 251.5, "completions/min_terminated_length": 21.1, "entropy": 0.26000291779637336, "epoch": 0.021, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 5.819999999999999e-07, "loss": 0.0, "num_tokens": 592882.0, "reward": 0.0, "reward_std": 0.0, "rewards/_reward/mean": 0.0, "rewards/_reward/std": 0.0, "step": 210, "step_time": 13.310213024541735 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.23193855732679367, "epoch": 0.022, "frac_reward_zero_std": 0.7, "grad_norm": 0.0, "learning_rate": 5.620000000000001e-07, "loss": -1.4901161193847657e-09, "num_tokens": 623314.0, "reward": 0.0625, "reward_std": 0.12246559858322144, "rewards/_reward/mean": 0.0625, "rewards/_reward/std": 0.12246559858322144, "step": 220, "step_time": 12.41725982457865 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.19535631276667118, "epoch": 0.023, "frac_reward_zero_std": 0.9, "grad_norm": 0.0, "learning_rate": 5.420000000000001e-07, "loss": -1.4901161193847657e-09, "num_tokens": 651706.0, "reward": 0.0125, "reward_std": 0.03535533845424652, "rewards/_reward/mean": 0.0125, "rewards/_reward/std": 0.03535533845424652, "step": 230, "step_time": 12.160462697478943 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.24304591715335847, "epoch": 0.024, "frac_reward_zero_std": 0.9, "grad_norm": 0.0, "learning_rate": 5.22e-07, "loss": -2.9802322387695314e-09, "num_tokens": 685298.0, "reward": 0.025, "reward_std": 0.046291005611419675, "rewards/_reward/mean": 0.025, "rewards/_reward/std": 0.046291005611419675, "step": 240, "step_time": 12.644157648505644 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.219215127825737, "epoch": 0.025, "frac_reward_zero_std": 0.7, "grad_norm": 0.0, "learning_rate": 5.02e-07, "loss": 1.4901161193847657e-09, "num_tokens": 713786.0, "reward": 0.1375, "reward_std": 0.12246559858322144, "rewards/_reward/mean": 0.1375, "rewards/_reward/std": 0.12246559858322144, "step": 250, "step_time": 12.234266605251468 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.95, "completions/max_length": 256.0, "completions/max_terminated_length": 24.6, "completions/mean_length": 254.175, "completions/mean_terminated_length": 21.95, "completions/min_length": 248.0, "completions/min_terminated_length": 17.6, "entropy": 0.23214468434453012, "epoch": 0.026, "frac_reward_zero_std": 0.8, "grad_norm": 0.0, "learning_rate": 4.82e-07, "loss": -4.470348358154297e-09, "num_tokens": 741568.0, "reward": 0.0375, "reward_std": 0.0816463440656662, "rewards/_reward/mean": 0.0375, "rewards/_reward/std": 0.0816463440656662, "step": 260, "step_time": 11.861435349751265 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.3036452613770962, "epoch": 0.027, "frac_reward_zero_std": 0.9, "grad_norm": 0.0, "learning_rate": 4.62e-07, "loss": 0.0, "num_tokens": 769408.0, "reward": 0.05, "reward_std": 0.05345224738121033, "rewards/_reward/mean": 0.05, "rewards/_reward/std": 0.05345224738121033, "step": 270, "step_time": 11.96508414738346 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.28062224201858044, "epoch": 0.028, "frac_reward_zero_std": 0.9, "grad_norm": 0.0, "learning_rate": 4.4199999999999996e-07, "loss": -1.4901161193847657e-09, "num_tokens": 796792.0, "reward": 0.0125, "reward_std": 0.03535533845424652, "rewards/_reward/mean": 0.0125, "rewards/_reward/std": 0.03535533845424652, "step": 280, "step_time": 12.066837795078754 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.3000262677669525, "epoch": 0.029, "frac_reward_zero_std": 0.8, "grad_norm": 0.0, "learning_rate": 4.2199999999999994e-07, "loss": -1.4901161193847657e-09, "num_tokens": 822888.0, "reward": 0.0625, "reward_std": 0.08880758583545685, "rewards/_reward/mean": 0.0625, "rewards/_reward/std": 0.08880758583545685, "step": 290, "step_time": 11.958969497331418 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.2139466732740402, "epoch": 0.03, "frac_reward_zero_std": 0.9, "grad_norm": 0.0, "learning_rate": 4.02e-07, "loss": -1.4901161193847657e-09, "num_tokens": 851112.0, "reward": 0.0125, "reward_std": 0.03535533845424652, "rewards/_reward/mean": 0.0125, "rewards/_reward/std": 0.03535533845424652, "step": 300, "step_time": 12.465268377074972 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.21100819781422614, "epoch": 0.031, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 3.82e-07, "loss": 0.0, "num_tokens": 880296.0, "reward": 0.0, "reward_std": 0.0, "rewards/_reward/mean": 0.0, "rewards/_reward/std": 0.0, "step": 310, "step_time": 12.395630138413981 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.1856917714700103, "epoch": 0.032, "frac_reward_zero_std": 0.8, "grad_norm": 0.0, "learning_rate": 3.62e-07, "loss": -4.470348358154297e-09, "num_tokens": 909568.0, "reward": 0.0375, "reward_std": 0.0816463440656662, "rewards/_reward/mean": 0.0375, "rewards/_reward/std": 0.0816463440656662, "step": 320, "step_time": 12.158450052631087 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9875, "completions/max_length": 256.0, "completions/max_terminated_length": 22.2, "completions/mean_length": 255.575, "completions/mean_terminated_length": 22.2, "completions/min_length": 252.6, "completions/min_terminated_length": 22.2, "entropy": 0.23710842803120613, "epoch": 0.033, "frac_reward_zero_std": 0.7, "grad_norm": 0.0, "learning_rate": 3.42e-07, "loss": 0.004176859557628631, "num_tokens": 937678.0, "reward": 0.2, "reward_std": 0.11700168251991272, "rewards/_reward/mean": 0.2, "rewards/_reward/std": 0.11700168251991272, "step": 330, "step_time": 11.860222824080846 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9375, "completions/max_length": 256.0, "completions/max_terminated_length": 23.6, "completions/mean_length": 253.8125, "completions/mean_terminated_length": 22.1, "completions/min_length": 251.1, "completions/min_terminated_length": 20.7, "entropy": 0.26537639163434507, "epoch": 0.034, "frac_reward_zero_std": 0.8, "grad_norm": 0.0, "learning_rate": 3.22e-07, "loss": 0.00504487007856369, "num_tokens": 967055.0, "reward": 0.1125, "reward_std": 0.09804592728614807, "rewards/_reward/mean": 0.1125, "rewards/_reward/std": 0.09804592728614807, "step": 340, "step_time": 11.909155616955832 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9875, "completions/max_length": 256.0, "completions/max_terminated_length": 14.8, "completions/mean_length": 254.65, "completions/mean_terminated_length": 14.8, "completions/min_length": 245.2, "completions/min_terminated_length": 14.8, "entropy": 0.2996050529181957, "epoch": 0.035, "frac_reward_zero_std": 0.9, "grad_norm": 0.0, "learning_rate": 3.02e-07, "loss": -2.9802322387695314e-09, "num_tokens": 995547.0, "reward": 0.025, "reward_std": 0.046291005611419675, "rewards/_reward/mean": 0.025, "rewards/_reward/std": 0.046291005611419675, "step": 350, "step_time": 11.996557193179616 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.32317615337669847, "epoch": 0.036, "frac_reward_zero_std": 0.5, "grad_norm": 0.0, "learning_rate": 2.8199999999999996e-07, "loss": -2.9802322387695314e-09, "num_tokens": 1023283.0, "reward": 0.125, "reward_std": 0.2205115258693695, "rewards/_reward/mean": 0.125, "rewards/_reward/std": 0.2205115258693695, "step": 360, "step_time": 12.106001363391988 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.24275533333420754, "epoch": 0.037, "frac_reward_zero_std": 0.8, "grad_norm": 0.0, "learning_rate": 2.62e-07, "loss": -2.9802322387695314e-09, "num_tokens": 1052363.0, "reward": 0.075, "reward_std": 0.09974325299263001, "rewards/_reward/mean": 0.075, "rewards/_reward/std": 0.09974325299263001, "step": 370, "step_time": 12.040205320669338 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.24013530761003493, "epoch": 0.038, "frac_reward_zero_std": 0.9, "grad_norm": 0.0, "learning_rate": 2.4199999999999997e-07, "loss": 0.0, "num_tokens": 1082899.0, "reward": 0.05, "reward_std": 0.05345224738121033, "rewards/_reward/mean": 0.05, "rewards/_reward/std": 0.05345224738121033, "step": 380, "step_time": 12.06431267345324 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.2428251437842846, "epoch": 0.039, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.22e-07, "loss": 0.0, "num_tokens": 1110755.0, "reward": 0.0, "reward_std": 0.0, "rewards/_reward/mean": 0.0, "rewards/_reward/std": 0.0, "step": 390, "step_time": 11.996180486050434 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.23162611648440362, "epoch": 0.04, "frac_reward_zero_std": 0.9, "grad_norm": 0.0, "learning_rate": 2.02e-07, "loss": -1.4901161193847657e-09, "num_tokens": 1140787.0, "reward": 0.0125, "reward_std": 0.03535533845424652, "rewards/_reward/mean": 0.0125, "rewards/_reward/std": 0.03535533845424652, "step": 400, "step_time": 12.043787954607978 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.19889585115015507, "epoch": 0.041, "frac_reward_zero_std": 0.9, "grad_norm": 3.5882863998413086, "learning_rate": 1.82e-07, "loss": -2.9802322387695314e-09, "num_tokens": 1171195.0, "reward": 0.025, "reward_std": 0.046291005611419675, "rewards/_reward/mean": 0.025, "rewards/_reward/std": 0.046291005611419675, "step": 410, "step_time": 12.544210581574589 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9875, "completions/max_length": 256.0, "completions/max_terminated_length": 21.1, "completions/mean_length": 255.4375, "completions/mean_terminated_length": 21.1, "completions/min_length": 251.5, "completions/min_terminated_length": 21.1, "entropy": 0.2701777949929237, "epoch": 0.042, "frac_reward_zero_std": 0.9, "grad_norm": 0.0, "learning_rate": 1.62e-07, "loss": -1.4901161193847657e-09, "num_tokens": 1200286.0, "reward": 0.0125, "reward_std": 0.03535533845424652, "rewards/_reward/mean": 0.0125, "rewards/_reward/std": 0.03535533845424652, "step": 420, "step_time": 12.16728035644628 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.3225971393287182, "epoch": 0.043, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.4199999999999997e-07, "loss": 0.0, "num_tokens": 1227534.0, "reward": 0.0, "reward_std": 0.0, "rewards/_reward/mean": 0.0, "rewards/_reward/std": 0.0, "step": 430, "step_time": 11.870932286046445 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9875, "completions/max_length": 256.0, "completions/max_terminated_length": 21.6, "completions/mean_length": 255.5, "completions/mean_terminated_length": 21.6, "completions/min_length": 252.0, "completions/min_terminated_length": 21.6, "entropy": 0.31563936099410056, "epoch": 0.044, "frac_reward_zero_std": 0.9, "grad_norm": 0.0, "learning_rate": 1.2199999999999998e-07, "loss": 0.0018630266189575196, "num_tokens": 1255606.0, "reward": 0.05, "reward_std": 0.05345224738121033, "rewards/_reward/mean": 0.05, "rewards/_reward/std": 0.05345224738121033, "step": 440, "step_time": 11.937281881668605 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.29385386109352113, "epoch": 0.045, "frac_reward_zero_std": 0.9, "grad_norm": 0.0, "learning_rate": 1.0199999999999999e-07, "loss": -1.4901161193847657e-09, "num_tokens": 1283918.0, "reward": 0.0125, "reward_std": 0.03535533845424652, "rewards/_reward/mean": 0.0125, "rewards/_reward/std": 0.03535533845424652, "step": 450, "step_time": 11.907829734543338 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.24251684471964835, "epoch": 0.046, "frac_reward_zero_std": 0.8, "grad_norm": 2.7865607738494873, "learning_rate": 8.2e-08, "loss": -2.9802322387695314e-09, "num_tokens": 1311366.0, "reward": 0.075, "reward_std": 0.08711026012897491, "rewards/_reward/mean": 0.075, "rewards/_reward/std": 0.08711026012897491, "step": 460, "step_time": 11.847331281122752 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.26004682183265687, "epoch": 0.047, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 6.2e-08, "loss": 0.0, "num_tokens": 1340998.0, "reward": 0.0, "reward_std": 0.0, "rewards/_reward/mean": 0.0, "rewards/_reward/std": 0.0, "step": 470, "step_time": 11.923272962146438 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.26978809721767905, "epoch": 0.048, "frac_reward_zero_std": 0.9, "grad_norm": 0.0, "learning_rate": 4.2e-08, "loss": 2.9802322387695314e-09, "num_tokens": 1367454.0, "reward": 0.075, "reward_std": 0.046291005611419675, "rewards/_reward/mean": 0.075, "rewards/_reward/std": 0.046291005611419675, "step": 480, "step_time": 11.881628181389534 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.2236021153628826, "epoch": 0.049, "frac_reward_zero_std": 0.8, "grad_norm": 0.0, "learning_rate": 2.2e-08, "loss": -4.470348358154297e-09, "num_tokens": 1398310.0, "reward": 0.0375, "reward_std": 0.0816463440656662, "rewards/_reward/mean": 0.0375, "rewards/_reward/std": 0.0816463440656662, "step": 490, "step_time": 11.971941134915687 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "entropy": 0.33287600688636304, "epoch": 0.05, "frac_reward_zero_std": 0.8, "grad_norm": 0.0, "learning_rate": 2e-09, "loss": -2.9802322387695314e-09, "num_tokens": 1425470.0, "reward": 0.025, "reward_std": 0.07071067690849304, "rewards/_reward/mean": 0.025, "rewards/_reward/std": 0.07071067690849304, "step": 500, "step_time": 11.938134964392521 } ], "logging_steps": 10, "max_steps": 500, "num_input_tokens_seen": 1425470, "num_train_epochs": 1, "save_steps": 100, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 0.0, "train_batch_size": 4, "trial_name": null, "trial_params": null }