Files
occ-grpo-baseline/last-checkpoint/trainer_state.json

1412 lines
46 KiB
JSON
Raw Normal View History

{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 0.05,
"eval_steps": 500,
"global_step": 500,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.14998279511928558,
"epoch": 0.0001,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0,
"learning_rate": 1e-06,
"loss": 0.0,
"num_tokens": 2568.0,
"reward": 0.0,
"reward_std": 0.0,
"rewards/_reward/mean": 0.0,
"rewards/_reward/std": 0.0,
"step": 1,
"step_time": 12.454096379224211
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.3256736550894048,
"epoch": 0.001,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0,
"learning_rate": 9.819999999999999e-07,
"loss": 0.0,
"num_tokens": 27464.0,
"reward": 0.0,
"reward_std": 0.0,
"rewards/_reward/mean": 0.0,
"rewards/_reward/std": 0.0,
"step": 10,
"step_time": 11.780018932263678
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.2880026239901781,
"epoch": 0.002,
"frac_reward_zero_std": 0.8,
"grad_norm": 0.0,
"learning_rate": 9.619999999999999e-07,
"loss": 0.0,
"num_tokens": 55576.0,
"reward": 0.1,
"reward_std": 0.09258201122283935,
"rewards/_reward/mean": 0.1,
"rewards/_reward/std": 0.09258201122283935,
"step": 20,
"step_time": 12.011081893160007
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.9875,
"completions/max_length": 256.0,
"completions/max_terminated_length": 17.5,
"completions/mean_length": 254.9875,
"completions/mean_terminated_length": 17.5,
"completions/min_length": 247.9,
"completions/min_terminated_length": 17.5,
"entropy": 0.28849115688353777,
"epoch": 0.003,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0,
"learning_rate": 9.419999999999999e-07,
"loss": 0.0,
"num_tokens": 83367.0,
"reward": 0.0,
"reward_std": 0.0,
"rewards/_reward/mean": 0.0,
"rewards/_reward/std": 0.0,
"step": 30,
"step_time": 12.034194040577859
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.95,
"completions/max_length": 256.0,
"completions/max_terminated_length": 23.4,
"completions/mean_length": 253.9,
"completions/mean_terminated_length": 21.4,
"completions/min_length": 249.1,
"completions/min_terminated_length": 18.7,
"entropy": 0.2561770185828209,
"epoch": 0.004,
"frac_reward_zero_std": 0.9,
"grad_norm": 2.998258590698242,
"learning_rate": 9.22e-07,
"loss": 1.4901161193847657e-09,
"num_tokens": 111439.0,
"reward": 0.0875,
"reward_std": 0.03535533845424652,
"rewards/_reward/mean": 0.0875,
"rewards/_reward/std": 0.03535533845424652,
"step": 40,
"step_time": 12.092716509709135
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.21435276679694654,
"epoch": 0.005,
"frac_reward_zero_std": 0.9,
"grad_norm": 0.0,
"learning_rate": 9.02e-07,
"loss": 1.4901161193847657e-09,
"num_tokens": 139943.0,
"reward": 0.0375,
"reward_std": 0.051754921674728394,
"rewards/_reward/mean": 0.0375,
"rewards/_reward/std": 0.051754921674728394,
"step": 50,
"step_time": 11.898367898073047
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.20879782736301422,
"epoch": 0.006,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0,
"learning_rate": 8.82e-07,
"loss": 0.0,
"num_tokens": 168911.0,
"reward": 0.0,
"reward_std": 0.0,
"rewards/_reward/mean": 0.0,
"rewards/_reward/std": 0.0,
"step": 60,
"step_time": 11.92987802445423
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.23100027367472648,
"epoch": 0.007,
"frac_reward_zero_std": 0.9,
"grad_norm": 0.0,
"learning_rate": 8.62e-07,
"loss": -1.4901161193847657e-09,
"num_tokens": 197359.0,
"reward": 0.0125,
"reward_std": 0.03535533845424652,
"rewards/_reward/mean": 0.0125,
"rewards/_reward/std": 0.03535533845424652,
"step": 70,
"step_time": 11.958151014475153
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.24868577010929585,
"epoch": 0.008,
"frac_reward_zero_std": 0.8,
"grad_norm": 2.808773994445801,
"learning_rate": 8.419999999999999e-07,
"loss": 0.0,
"num_tokens": 224775.0,
"reward": 0.15,
"reward_std": 0.08711026012897491,
"rewards/_reward/mean": 0.15,
"rewards/_reward/std": 0.08711026012897491,
"step": 80,
"step_time": 11.96262693060562
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.23981668353080748,
"epoch": 0.009,
"frac_reward_zero_std": 0.9,
"grad_norm": 0.0,
"learning_rate": 8.219999999999999e-07,
"loss": -1.4901161193847657e-09,
"num_tokens": 252687.0,
"reward": 0.0625,
"reward_std": 0.051754921674728394,
"rewards/_reward/mean": 0.0625,
"rewards/_reward/std": 0.051754921674728394,
"step": 90,
"step_time": 11.79044756719377
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.9875,
"completions/max_length": 256.0,
"completions/max_terminated_length": 25.6,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 25.6,
"completions/min_length": 256.0,
"completions/min_terminated_length": 25.6,
"entropy": 0.2731012573465705,
"epoch": 0.01,
"frac_reward_zero_std": 0.9,
"grad_norm": 0.0,
"learning_rate": 8.02e-07,
"loss": 0.0,
"num_tokens": 280847.0,
"reward": 0.0625,
"reward_std": 0.051754921674728394,
"rewards/_reward/mean": 0.0625,
"rewards/_reward/std": 0.051754921674728394,
"step": 100,
"step_time": 11.873896015947684
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.2539512518793344,
"epoch": 0.011,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0,
"learning_rate": 7.82e-07,
"loss": 0.0,
"num_tokens": 308583.0,
"reward": 0.0,
"reward_std": 0.0,
"rewards/_reward/mean": 0.0,
"rewards/_reward/std": 0.0,
"step": 110,
"step_time": 12.493365852814168
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.23124769441783427,
"epoch": 0.012,
"frac_reward_zero_std": 0.8,
"grad_norm": 0.0,
"learning_rate": 7.62e-07,
"loss": -4.470348358154297e-09,
"num_tokens": 336751.0,
"reward": 0.0375,
"reward_std": 0.0816463440656662,
"rewards/_reward/mean": 0.0375,
"rewards/_reward/std": 0.0816463440656662,
"step": 120,
"step_time": 12.216748453862966
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.24605347998440266,
"epoch": 0.013,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0,
"learning_rate": 7.42e-07,
"loss": 0.0,
"num_tokens": 366007.0,
"reward": 0.0,
"reward_std": 0.0,
"rewards/_reward/mean": 0.0,
"rewards/_reward/std": 0.0,
"step": 130,
"step_time": 11.979620195319876
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.23477237336337567,
"epoch": 0.014,
"frac_reward_zero_std": 0.9,
"grad_norm": 0.0,
"learning_rate": 7.219999999999999e-07,
"loss": -1.4901161193847657e-09,
"num_tokens": 396623.0,
"reward": 0.0125,
"reward_std": 0.03535533845424652,
"rewards/_reward/mean": 0.0125,
"rewards/_reward/std": 0.03535533845424652,
"step": 140,
"step_time": 11.984532529045827
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.19286549724638463,
"epoch": 0.015,
"frac_reward_zero_std": 0.7,
"grad_norm": 0.0,
"learning_rate": 7.019999999999999e-07,
"loss": 1.4901161193847657e-09,
"num_tokens": 425055.0,
"reward": 0.0875,
"reward_std": 0.1388651818037033,
"rewards/_reward/mean": 0.0875,
"rewards/_reward/std": 0.1388651818037033,
"step": 150,
"step_time": 11.88744165727403
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.2272243991494179,
"epoch": 0.016,
"frac_reward_zero_std": 0.7,
"grad_norm": 3.0967137813568115,
"learning_rate": 6.82e-07,
"loss": 0.0,
"num_tokens": 453151.0,
"reward": 0.15,
"reward_std": 0.1334012657403946,
"rewards/_reward/mean": 0.15,
"rewards/_reward/std": 0.1334012657403946,
"step": 160,
"step_time": 12.118341535096988
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.19670370109379293,
"epoch": 0.017,
"frac_reward_zero_std": 0.8,
"grad_norm": 0.0,
"learning_rate": 6.62e-07,
"loss": 0.0,
"num_tokens": 480895.0,
"reward": 0.05,
"reward_std": 0.08711026012897491,
"rewards/_reward/mean": 0.05,
"rewards/_reward/std": 0.08711026012897491,
"step": 170,
"step_time": 11.987511804816313
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.28401327803730964,
"epoch": 0.018,
"frac_reward_zero_std": 0.9,
"grad_norm": 0.0,
"learning_rate": 6.42e-07,
"loss": -1.4901161193847657e-09,
"num_tokens": 508823.0,
"reward": 0.0625,
"reward_std": 0.051754921674728394,
"rewards/_reward/mean": 0.0625,
"rewards/_reward/std": 0.051754921674728394,
"step": 180,
"step_time": 11.97194695000071
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.2276454884558916,
"epoch": 0.019,
"frac_reward_zero_std": 0.6,
"grad_norm": 0.0,
"learning_rate": 6.219999999999999e-07,
"loss": -5.960464477539063e-09,
"num_tokens": 537039.0,
"reward": 0.075,
"reward_std": 0.1632926881313324,
"rewards/_reward/mean": 0.075,
"rewards/_reward/std": 0.1632926881313324,
"step": 190,
"step_time": 11.945305320294574
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.21875664182007312,
"epoch": 0.02,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0,
"learning_rate": 6.019999999999999e-07,
"loss": 0.0,
"num_tokens": 565071.0,
"reward": 0.0,
"reward_std": 0.0,
"rewards/_reward/mean": 0.0,
"rewards/_reward/std": 0.0,
"step": 200,
"step_time": 12.232728651817888
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.9875,
"completions/max_length": 256.0,
"completions/max_terminated_length": 21.1,
"completions/mean_length": 255.4375,
"completions/mean_terminated_length": 21.1,
"completions/min_length": 251.5,
"completions/min_terminated_length": 21.1,
"entropy": 0.26000291779637336,
"epoch": 0.021,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0,
"learning_rate": 5.819999999999999e-07,
"loss": 0.0,
"num_tokens": 592882.0,
"reward": 0.0,
"reward_std": 0.0,
"rewards/_reward/mean": 0.0,
"rewards/_reward/std": 0.0,
"step": 210,
"step_time": 13.310213024541735
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.23193855732679367,
"epoch": 0.022,
"frac_reward_zero_std": 0.7,
"grad_norm": 0.0,
"learning_rate": 5.620000000000001e-07,
"loss": -1.4901161193847657e-09,
"num_tokens": 623314.0,
"reward": 0.0625,
"reward_std": 0.12246559858322144,
"rewards/_reward/mean": 0.0625,
"rewards/_reward/std": 0.12246559858322144,
"step": 220,
"step_time": 12.41725982457865
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.19535631276667118,
"epoch": 0.023,
"frac_reward_zero_std": 0.9,
"grad_norm": 0.0,
"learning_rate": 5.420000000000001e-07,
"loss": -1.4901161193847657e-09,
"num_tokens": 651706.0,
"reward": 0.0125,
"reward_std": 0.03535533845424652,
"rewards/_reward/mean": 0.0125,
"rewards/_reward/std": 0.03535533845424652,
"step": 230,
"step_time": 12.160462697478943
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.24304591715335847,
"epoch": 0.024,
"frac_reward_zero_std": 0.9,
"grad_norm": 0.0,
"learning_rate": 5.22e-07,
"loss": -2.9802322387695314e-09,
"num_tokens": 685298.0,
"reward": 0.025,
"reward_std": 0.046291005611419675,
"rewards/_reward/mean": 0.025,
"rewards/_reward/std": 0.046291005611419675,
"step": 240,
"step_time": 12.644157648505644
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.219215127825737,
"epoch": 0.025,
"frac_reward_zero_std": 0.7,
"grad_norm": 0.0,
"learning_rate": 5.02e-07,
"loss": 1.4901161193847657e-09,
"num_tokens": 713786.0,
"reward": 0.1375,
"reward_std": 0.12246559858322144,
"rewards/_reward/mean": 0.1375,
"rewards/_reward/std": 0.12246559858322144,
"step": 250,
"step_time": 12.234266605251468
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.95,
"completions/max_length": 256.0,
"completions/max_terminated_length": 24.6,
"completions/mean_length": 254.175,
"completions/mean_terminated_length": 21.95,
"completions/min_length": 248.0,
"completions/min_terminated_length": 17.6,
"entropy": 0.23214468434453012,
"epoch": 0.026,
"frac_reward_zero_std": 0.8,
"grad_norm": 0.0,
"learning_rate": 4.82e-07,
"loss": -4.470348358154297e-09,
"num_tokens": 741568.0,
"reward": 0.0375,
"reward_std": 0.0816463440656662,
"rewards/_reward/mean": 0.0375,
"rewards/_reward/std": 0.0816463440656662,
"step": 260,
"step_time": 11.861435349751265
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.3036452613770962,
"epoch": 0.027,
"frac_reward_zero_std": 0.9,
"grad_norm": 0.0,
"learning_rate": 4.62e-07,
"loss": 0.0,
"num_tokens": 769408.0,
"reward": 0.05,
"reward_std": 0.05345224738121033,
"rewards/_reward/mean": 0.05,
"rewards/_reward/std": 0.05345224738121033,
"step": 270,
"step_time": 11.96508414738346
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.28062224201858044,
"epoch": 0.028,
"frac_reward_zero_std": 0.9,
"grad_norm": 0.0,
"learning_rate": 4.4199999999999996e-07,
"loss": -1.4901161193847657e-09,
"num_tokens": 796792.0,
"reward": 0.0125,
"reward_std": 0.03535533845424652,
"rewards/_reward/mean": 0.0125,
"rewards/_reward/std": 0.03535533845424652,
"step": 280,
"step_time": 12.066837795078754
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.3000262677669525,
"epoch": 0.029,
"frac_reward_zero_std": 0.8,
"grad_norm": 0.0,
"learning_rate": 4.2199999999999994e-07,
"loss": -1.4901161193847657e-09,
"num_tokens": 822888.0,
"reward": 0.0625,
"reward_std": 0.08880758583545685,
"rewards/_reward/mean": 0.0625,
"rewards/_reward/std": 0.08880758583545685,
"step": 290,
"step_time": 11.958969497331418
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.2139466732740402,
"epoch": 0.03,
"frac_reward_zero_std": 0.9,
"grad_norm": 0.0,
"learning_rate": 4.02e-07,
"loss": -1.4901161193847657e-09,
"num_tokens": 851112.0,
"reward": 0.0125,
"reward_std": 0.03535533845424652,
"rewards/_reward/mean": 0.0125,
"rewards/_reward/std": 0.03535533845424652,
"step": 300,
"step_time": 12.465268377074972
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.21100819781422614,
"epoch": 0.031,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0,
"learning_rate": 3.82e-07,
"loss": 0.0,
"num_tokens": 880296.0,
"reward": 0.0,
"reward_std": 0.0,
"rewards/_reward/mean": 0.0,
"rewards/_reward/std": 0.0,
"step": 310,
"step_time": 12.395630138413981
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.1856917714700103,
"epoch": 0.032,
"frac_reward_zero_std": 0.8,
"grad_norm": 0.0,
"learning_rate": 3.62e-07,
"loss": -4.470348358154297e-09,
"num_tokens": 909568.0,
"reward": 0.0375,
"reward_std": 0.0816463440656662,
"rewards/_reward/mean": 0.0375,
"rewards/_reward/std": 0.0816463440656662,
"step": 320,
"step_time": 12.158450052631087
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.9875,
"completions/max_length": 256.0,
"completions/max_terminated_length": 22.2,
"completions/mean_length": 255.575,
"completions/mean_terminated_length": 22.2,
"completions/min_length": 252.6,
"completions/min_terminated_length": 22.2,
"entropy": 0.23710842803120613,
"epoch": 0.033,
"frac_reward_zero_std": 0.7,
"grad_norm": 0.0,
"learning_rate": 3.42e-07,
"loss": 0.004176859557628631,
"num_tokens": 937678.0,
"reward": 0.2,
"reward_std": 0.11700168251991272,
"rewards/_reward/mean": 0.2,
"rewards/_reward/std": 0.11700168251991272,
"step": 330,
"step_time": 11.860222824080846
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.9375,
"completions/max_length": 256.0,
"completions/max_terminated_length": 23.6,
"completions/mean_length": 253.8125,
"completions/mean_terminated_length": 22.1,
"completions/min_length": 251.1,
"completions/min_terminated_length": 20.7,
"entropy": 0.26537639163434507,
"epoch": 0.034,
"frac_reward_zero_std": 0.8,
"grad_norm": 0.0,
"learning_rate": 3.22e-07,
"loss": 0.00504487007856369,
"num_tokens": 967055.0,
"reward": 0.1125,
"reward_std": 0.09804592728614807,
"rewards/_reward/mean": 0.1125,
"rewards/_reward/std": 0.09804592728614807,
"step": 340,
"step_time": 11.909155616955832
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.9875,
"completions/max_length": 256.0,
"completions/max_terminated_length": 14.8,
"completions/mean_length": 254.65,
"completions/mean_terminated_length": 14.8,
"completions/min_length": 245.2,
"completions/min_terminated_length": 14.8,
"entropy": 0.2996050529181957,
"epoch": 0.035,
"frac_reward_zero_std": 0.9,
"grad_norm": 0.0,
"learning_rate": 3.02e-07,
"loss": -2.9802322387695314e-09,
"num_tokens": 995547.0,
"reward": 0.025,
"reward_std": 0.046291005611419675,
"rewards/_reward/mean": 0.025,
"rewards/_reward/std": 0.046291005611419675,
"step": 350,
"step_time": 11.996557193179616
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.32317615337669847,
"epoch": 0.036,
"frac_reward_zero_std": 0.5,
"grad_norm": 0.0,
"learning_rate": 2.8199999999999996e-07,
"loss": -2.9802322387695314e-09,
"num_tokens": 1023283.0,
"reward": 0.125,
"reward_std": 0.2205115258693695,
"rewards/_reward/mean": 0.125,
"rewards/_reward/std": 0.2205115258693695,
"step": 360,
"step_time": 12.106001363391988
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.24275533333420754,
"epoch": 0.037,
"frac_reward_zero_std": 0.8,
"grad_norm": 0.0,
"learning_rate": 2.62e-07,
"loss": -2.9802322387695314e-09,
"num_tokens": 1052363.0,
"reward": 0.075,
"reward_std": 0.09974325299263001,
"rewards/_reward/mean": 0.075,
"rewards/_reward/std": 0.09974325299263001,
"step": 370,
"step_time": 12.040205320669338
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.24013530761003493,
"epoch": 0.038,
"frac_reward_zero_std": 0.9,
"grad_norm": 0.0,
"learning_rate": 2.4199999999999997e-07,
"loss": 0.0,
"num_tokens": 1082899.0,
"reward": 0.05,
"reward_std": 0.05345224738121033,
"rewards/_reward/mean": 0.05,
"rewards/_reward/std": 0.05345224738121033,
"step": 380,
"step_time": 12.06431267345324
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.2428251437842846,
"epoch": 0.039,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0,
"learning_rate": 2.22e-07,
"loss": 0.0,
"num_tokens": 1110755.0,
"reward": 0.0,
"reward_std": 0.0,
"rewards/_reward/mean": 0.0,
"rewards/_reward/std": 0.0,
"step": 390,
"step_time": 11.996180486050434
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.23162611648440362,
"epoch": 0.04,
"frac_reward_zero_std": 0.9,
"grad_norm": 0.0,
"learning_rate": 2.02e-07,
"loss": -1.4901161193847657e-09,
"num_tokens": 1140787.0,
"reward": 0.0125,
"reward_std": 0.03535533845424652,
"rewards/_reward/mean": 0.0125,
"rewards/_reward/std": 0.03535533845424652,
"step": 400,
"step_time": 12.043787954607978
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.19889585115015507,
"epoch": 0.041,
"frac_reward_zero_std": 0.9,
"grad_norm": 3.5882863998413086,
"learning_rate": 1.82e-07,
"loss": -2.9802322387695314e-09,
"num_tokens": 1171195.0,
"reward": 0.025,
"reward_std": 0.046291005611419675,
"rewards/_reward/mean": 0.025,
"rewards/_reward/std": 0.046291005611419675,
"step": 410,
"step_time": 12.544210581574589
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.9875,
"completions/max_length": 256.0,
"completions/max_terminated_length": 21.1,
"completions/mean_length": 255.4375,
"completions/mean_terminated_length": 21.1,
"completions/min_length": 251.5,
"completions/min_terminated_length": 21.1,
"entropy": 0.2701777949929237,
"epoch": 0.042,
"frac_reward_zero_std": 0.9,
"grad_norm": 0.0,
"learning_rate": 1.62e-07,
"loss": -1.4901161193847657e-09,
"num_tokens": 1200286.0,
"reward": 0.0125,
"reward_std": 0.03535533845424652,
"rewards/_reward/mean": 0.0125,
"rewards/_reward/std": 0.03535533845424652,
"step": 420,
"step_time": 12.16728035644628
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.3225971393287182,
"epoch": 0.043,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0,
"learning_rate": 1.4199999999999997e-07,
"loss": 0.0,
"num_tokens": 1227534.0,
"reward": 0.0,
"reward_std": 0.0,
"rewards/_reward/mean": 0.0,
"rewards/_reward/std": 0.0,
"step": 430,
"step_time": 11.870932286046445
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.9875,
"completions/max_length": 256.0,
"completions/max_terminated_length": 21.6,
"completions/mean_length": 255.5,
"completions/mean_terminated_length": 21.6,
"completions/min_length": 252.0,
"completions/min_terminated_length": 21.6,
"entropy": 0.31563936099410056,
"epoch": 0.044,
"frac_reward_zero_std": 0.9,
"grad_norm": 0.0,
"learning_rate": 1.2199999999999998e-07,
"loss": 0.0018630266189575196,
"num_tokens": 1255606.0,
"reward": 0.05,
"reward_std": 0.05345224738121033,
"rewards/_reward/mean": 0.05,
"rewards/_reward/std": 0.05345224738121033,
"step": 440,
"step_time": 11.937281881668605
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.29385386109352113,
"epoch": 0.045,
"frac_reward_zero_std": 0.9,
"grad_norm": 0.0,
"learning_rate": 1.0199999999999999e-07,
"loss": -1.4901161193847657e-09,
"num_tokens": 1283918.0,
"reward": 0.0125,
"reward_std": 0.03535533845424652,
"rewards/_reward/mean": 0.0125,
"rewards/_reward/std": 0.03535533845424652,
"step": 450,
"step_time": 11.907829734543338
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.24251684471964835,
"epoch": 0.046,
"frac_reward_zero_std": 0.8,
"grad_norm": 2.7865607738494873,
"learning_rate": 8.2e-08,
"loss": -2.9802322387695314e-09,
"num_tokens": 1311366.0,
"reward": 0.075,
"reward_std": 0.08711026012897491,
"rewards/_reward/mean": 0.075,
"rewards/_reward/std": 0.08711026012897491,
"step": 460,
"step_time": 11.847331281122752
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.26004682183265687,
"epoch": 0.047,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0,
"learning_rate": 6.2e-08,
"loss": 0.0,
"num_tokens": 1340998.0,
"reward": 0.0,
"reward_std": 0.0,
"rewards/_reward/mean": 0.0,
"rewards/_reward/std": 0.0,
"step": 470,
"step_time": 11.923272962146438
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.26978809721767905,
"epoch": 0.048,
"frac_reward_zero_std": 0.9,
"grad_norm": 0.0,
"learning_rate": 4.2e-08,
"loss": 2.9802322387695314e-09,
"num_tokens": 1367454.0,
"reward": 0.075,
"reward_std": 0.046291005611419675,
"rewards/_reward/mean": 0.075,
"rewards/_reward/std": 0.046291005611419675,
"step": 480,
"step_time": 11.881628181389534
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.2236021153628826,
"epoch": 0.049,
"frac_reward_zero_std": 0.8,
"grad_norm": 0.0,
"learning_rate": 2.2e-08,
"loss": -4.470348358154297e-09,
"num_tokens": 1398310.0,
"reward": 0.0375,
"reward_std": 0.0816463440656662,
"rewards/_reward/mean": 0.0375,
"rewards/_reward/std": 0.0816463440656662,
"step": 490,
"step_time": 11.971941134915687
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 256.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 256.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 256.0,
"completions/min_terminated_length": 0.0,
"entropy": 0.33287600688636304,
"epoch": 0.05,
"frac_reward_zero_std": 0.8,
"grad_norm": 0.0,
"learning_rate": 2e-09,
"loss": -2.9802322387695314e-09,
"num_tokens": 1425470.0,
"reward": 0.025,
"reward_std": 0.07071067690849304,
"rewards/_reward/mean": 0.025,
"rewards/_reward/std": 0.07071067690849304,
"step": 500,
"step_time": 11.938134964392521
}
],
"logging_steps": 10,
"max_steps": 500,
"num_input_tokens_seen": 1425470,
"num_train_epochs": 1,
"save_steps": 100,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": true
},
"attributes": {}
}
},
"total_flos": 0.0,
"train_batch_size": 4,
"trial_name": null,
"trial_params": null
}