{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 1.0, "eval_steps": 500, "global_step": 1250, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 470.0, "completions/mean_length": 276.6875, "completions/mean_terminated_length": 184.60870361328125, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "epoch": 0.0008, "grad_norm": 3.820082187652588, "kl": 0.0003216266632080078, "learning_rate": 0.0, "loss": -0.1909, "num_tokens": 15006.0, "reward": -8.20977783203125, "reward_std": 4.4750261306762695, "rewards/rm_reward_func/mean": -8.20977783203125, "rewards/rm_reward_func/std": 7.660107135772705, "step": 1 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.4375, "completions/max_length": 512.0, "completions/max_terminated_length": 424.0, "completions/mean_length": 313.625, "completions/mean_terminated_length": 159.3333282470703, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "epoch": 0.0016, "grad_norm": 4.830166339874268, "kl": 0.0002695322036743164, "learning_rate": 1.5873015873015872e-08, "loss": 0.0694, "num_tokens": 28018.0, "reward": -11.44384765625, "reward_std": 2.6692183017730713, "rewards/rm_reward_func/mean": -11.44384765625, "rewards/rm_reward_func/std": 6.70808744430542, "step": 2 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40625, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 374.28125, "completions/mean_terminated_length": 280.0526428222656, "completions/min_length": 83.0, "completions/min_terminated_length": 83.0, "epoch": 0.0024, "grad_norm": 2.5839879512786865, "kl": 0.0003209114074707031, "learning_rate": 3.1746031746031744e-08, "loss": -0.1093, "num_tokens": 42787.0, "reward": -13.593017578125, "reward_std": 5.063211917877197, "rewards/rm_reward_func/mean": -13.593017578125, "rewards/rm_reward_func/std": 8.933624267578125, "step": 3 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 467.0, "completions/mean_length": 176.3125, "completions/mean_terminated_length": 141.58621215820312, "completions/min_length": 85.0, "completions/min_terminated_length": 85.0, "epoch": 0.0032, "grad_norm": 3.8852806091308594, "kl": 0.00033855438232421875, "learning_rate": 4.7619047619047613e-08, "loss": -0.0693, "num_tokens": 50509.0, "reward": -8.043212890625, "reward_std": 6.7915544509887695, "rewards/rm_reward_func/mean": -8.043212890625, "rewards/rm_reward_func/std": 9.116313934326172, "step": 4 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 450.0, "completions/mean_length": 240.59375, "completions/mean_terminated_length": 212.51724243164062, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "epoch": 0.004, "grad_norm": 2.737971544265747, "kl": 0.0002636909484863281, "learning_rate": 6.349206349206349e-08, "loss": -0.0144, "num_tokens": 61880.0, "reward": -7.558570861816406, "reward_std": 4.206568717956543, "rewards/rm_reward_func/mean": -7.558570861816406, "rewards/rm_reward_func/std": 6.464176177978516, "step": 5 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 449.0, "completions/mean_length": 268.5, "completions/mean_terminated_length": 187.33334350585938, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "epoch": 0.0048, "grad_norm": 2.704625368118286, "kl": 0.00022745132446289062, "learning_rate": 7.936507936507936e-08, "loss": -0.0148, "num_tokens": 73048.0, "reward": -8.5592041015625, "reward_std": 6.456365585327148, "rewards/rm_reward_func/mean": -8.5592041015625, "rewards/rm_reward_func/std": 6.686816692352295, "step": 6 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 308.0, "completions/mean_length": 255.34375, "completions/mean_terminated_length": 169.7916717529297, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "epoch": 0.0056, "grad_norm": 3.0464067459106445, "kl": 0.00023818016052246094, "learning_rate": 9.523809523809523e-08, "loss": 0.2187, "num_tokens": 84043.0, "reward": -10.86029052734375, "reward_std": 5.81391716003418, "rewards/rm_reward_func/mean": -10.86029052734375, "rewards/rm_reward_func/std": 7.67104959487915, "step": 7 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.46875, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 370.15625, "completions/mean_terminated_length": 245.0, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "epoch": 0.0064, "grad_norm": 2.1677191257476807, "kl": 0.0003154277801513672, "learning_rate": 1.111111111111111e-07, "loss": -0.0946, "num_tokens": 97960.0, "reward": -9.22998046875, "reward_std": 4.78300666809082, "rewards/rm_reward_func/mean": -9.22998046875, "rewards/rm_reward_func/std": 15.388681411743164, "step": 8 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 419.0, "completions/mean_length": 355.5, "completions/mean_terminated_length": 294.2608642578125, "completions/min_length": 181.0, "completions/min_terminated_length": 181.0, "epoch": 0.0072, "grad_norm": 2.2008018493652344, "kl": 0.00035500526428222656, "learning_rate": 1.2698412698412698e-07, "loss": 0.0485, "num_tokens": 113944.0, "reward": -7.944580078125, "reward_std": 4.6264190673828125, "rewards/rm_reward_func/mean": -7.944580078125, "rewards/rm_reward_func/std": 7.673216342926025, "step": 9 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 398.65625, "completions/mean_terminated_length": 285.3125, "completions/min_length": 135.0, "completions/min_terminated_length": 135.0, "epoch": 0.008, "grad_norm": 2.5219130516052246, "kl": 0.0003647804260253906, "learning_rate": 1.4285714285714285e-07, "loss": -0.0682, "num_tokens": 130357.0, "reward": -14.8125, "reward_std": 3.55131196975708, "rewards/rm_reward_func/mean": -14.8125, "rewards/rm_reward_func/std": 6.269670486450195, "step": 10 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.75, "completions/max_length": 512.0, "completions/max_terminated_length": 468.0, "completions/mean_length": 438.03125, "completions/mean_terminated_length": 216.125, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "epoch": 0.0088, "grad_norm": 2.1736643314361572, "kl": 0.00029921531677246094, "learning_rate": 1.5873015873015872e-07, "loss": 0.2198, "num_tokens": 148190.0, "reward": -13.016815185546875, "reward_std": 7.495743274688721, "rewards/rm_reward_func/mean": -13.016815185546875, "rewards/rm_reward_func/std": 8.30325698852539, "step": 11 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 356.0, "completions/mean_length": 178.5625, "completions/mean_terminated_length": 144.0689697265625, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "epoch": 0.0096, "grad_norm": 4.596688747406006, "kl": 0.00018262863159179688, "learning_rate": 1.7460317460317458e-07, "loss": 0.0833, "num_tokens": 156104.0, "reward": 0.974029541015625, "reward_std": 4.125612735748291, "rewards/rm_reward_func/mean": 0.974029541015625, "rewards/rm_reward_func/std": 9.185401916503906, "step": 12 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 404.0, "completions/max_terminated_length": 404.0, "completions/mean_length": 250.25, "completions/mean_terminated_length": 250.25, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "epoch": 0.0104, "grad_norm": 3.2415318489074707, "kl": 0.0003223419189453125, "learning_rate": 1.9047619047619045e-07, "loss": -0.1464, "num_tokens": 166208.0, "reward": -8.69970703125, "reward_std": 4.659298419952393, "rewards/rm_reward_func/mean": -8.69970703125, "rewards/rm_reward_func/std": 5.567142963409424, "step": 13 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 328.03125, "completions/mean_terminated_length": 244.4091033935547, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "epoch": 0.0112, "grad_norm": 2.257594347000122, "kl": 0.0003323554992675781, "learning_rate": 2.0634920634920632e-07, "loss": -0.0784, "num_tokens": 178833.0, "reward": -8.506607055664062, "reward_std": 7.695546627044678, "rewards/rm_reward_func/mean": -8.506607055664062, "rewards/rm_reward_func/std": 8.626959800720215, "step": 14 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 274.65625, "completions/mean_terminated_length": 150.33334350585938, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "epoch": 0.012, "grad_norm": 2.8894879817962646, "kl": 0.00014698505401611328, "learning_rate": 2.222222222222222e-07, "loss": 0.1296, "num_tokens": 192198.0, "reward": -8.6026611328125, "reward_std": 7.466651916503906, "rewards/rm_reward_func/mean": -8.6026611328125, "rewards/rm_reward_func/std": 8.928447723388672, "step": 15 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 431.0, "completions/max_terminated_length": 431.0, "completions/mean_length": 163.53125, "completions/mean_terminated_length": 163.53125, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "epoch": 0.0128, "grad_norm": 6.473876476287842, "kl": 0.00042629241943359375, "learning_rate": 2.3809523809523806e-07, "loss": -0.1249, "num_tokens": 203023.0, "reward": -5.33013916015625, "reward_std": 5.538287162780762, "rewards/rm_reward_func/mean": -5.33013916015625, "rewards/rm_reward_func/std": 8.5722074508667, "step": 16 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 325.0625, "completions/mean_terminated_length": 272.7200012207031, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "epoch": 0.0136, "grad_norm": 2.029867172241211, "kl": 0.0003304481506347656, "learning_rate": 2.5396825396825396e-07, "loss": 0.0048, "num_tokens": 215681.0, "reward": -8.299209594726562, "reward_std": 5.960148811340332, "rewards/rm_reward_func/mean": -8.299209594726562, "rewards/rm_reward_func/std": 9.681968688964844, "step": 17 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 467.0, "completions/mean_length": 309.46875, "completions/mean_terminated_length": 217.4091033935547, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "epoch": 0.0144, "grad_norm": 3.225783348083496, "kl": 0.0002518892288208008, "learning_rate": 2.698412698412698e-07, "loss": 0.454, "num_tokens": 228784.0, "reward": -12.326904296875, "reward_std": 6.084526062011719, "rewards/rm_reward_func/mean": -12.326904296875, "rewards/rm_reward_func/std": 10.45023250579834, "step": 18 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 246.84375, "completions/mean_terminated_length": 238.29031372070312, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "epoch": 0.0152, "grad_norm": 3.1062114238739014, "kl": 0.00031566619873046875, "learning_rate": 2.857142857142857e-07, "loss": -0.0427, "num_tokens": 238891.0, "reward": -5.867828369140625, "reward_std": 5.073713302612305, "rewards/rm_reward_func/mean": -5.867828369140625, "rewards/rm_reward_func/std": 6.714073181152344, "step": 19 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 463.0, "completions/mean_length": 279.0625, "completions/mean_terminated_length": 271.5483703613281, "completions/min_length": 19.0, "completions/min_terminated_length": 19.0, "epoch": 0.016, "grad_norm": 2.1619873046875, "kl": 0.00023698806762695312, "learning_rate": 3.0158730158730156e-07, "loss": -0.1689, "num_tokens": 249645.0, "reward": -3.2251663208007812, "reward_std": 4.3877034187316895, "rewards/rm_reward_func/mean": -3.2251663208007812, "rewards/rm_reward_func/std": 5.818446159362793, "step": 20 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 282.40625, "completions/mean_terminated_length": 258.6551818847656, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "epoch": 0.0168, "grad_norm": 4.3518900871276855, "kl": 0.0003197193145751953, "learning_rate": 3.1746031746031743e-07, "loss": 0.1906, "num_tokens": 261154.0, "reward": -8.756103515625, "reward_std": 3.641651153564453, "rewards/rm_reward_func/mean": -8.756103515625, "rewards/rm_reward_func/std": 4.865791320800781, "step": 21 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 256.125, "completions/mean_terminated_length": 184.47999572753906, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "epoch": 0.0176, "grad_norm": 3.306163787841797, "kl": 0.00031304359436035156, "learning_rate": 3.333333333333333e-07, "loss": 0.1021, "num_tokens": 272150.0, "reward": -13.5908203125, "reward_std": 7.745641708374023, "rewards/rm_reward_func/mean": -13.5908203125, "rewards/rm_reward_func/std": 8.889524459838867, "step": 22 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 436.0, "completions/mean_length": 355.59375, "completions/mean_terminated_length": 261.75, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "epoch": 0.0184, "grad_norm": 2.356691837310791, "kl": 0.0002713203430175781, "learning_rate": 3.4920634920634917e-07, "loss": -0.1426, "num_tokens": 286265.0, "reward": -8.7578125, "reward_std": 4.64816951751709, "rewards/rm_reward_func/mean": -8.7578125, "rewards/rm_reward_func/std": 10.985777854919434, "step": 23 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 238.65625, "completions/mean_terminated_length": 147.5416717529297, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "epoch": 0.0192, "grad_norm": 5.42078161239624, "kl": 0.0004291534423828125, "learning_rate": 3.6507936507936504e-07, "loss": 0.0164, "num_tokens": 297494.0, "reward": -6.644187927246094, "reward_std": 5.260914325714111, "rewards/rm_reward_func/mean": -6.644187927246094, "rewards/rm_reward_func/std": 10.8004150390625, "step": 24 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 481.0, "completions/mean_length": 231.4375, "completions/mean_terminated_length": 202.41378784179688, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "epoch": 0.02, "grad_norm": 5.109860420227051, "kl": 0.00030303001403808594, "learning_rate": 3.809523809523809e-07, "loss": 0.084, "num_tokens": 309052.0, "reward": -5.954833984375, "reward_std": 5.726484298706055, "rewards/rm_reward_func/mean": -5.954833984375, "rewards/rm_reward_func/std": 7.126110553741455, "step": 25 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 193.0, "completions/mean_length": 240.5625, "completions/mean_terminated_length": 77.70000457763672, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "epoch": 0.0208, "grad_norm": 3.4662835597991943, "kl": 0.00029218196868896484, "learning_rate": 3.968253968253968e-07, "loss": 0.1907, "num_tokens": 320334.0, "reward": -14.75677490234375, "reward_std": 4.248941421508789, "rewards/rm_reward_func/mean": -14.75677490234375, "rewards/rm_reward_func/std": 7.078120708465576, "step": 26 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.53125, "completions/max_length": 512.0, "completions/max_terminated_length": 492.0, "completions/mean_length": 450.09375, "completions/mean_terminated_length": 379.933349609375, "completions/min_length": 212.0, "completions/min_terminated_length": 212.0, "epoch": 0.0216, "grad_norm": 1.6936448812484741, "kl": 0.0002624988555908203, "learning_rate": 4.1269841269841265e-07, "loss": -0.0219, "num_tokens": 340865.0, "reward": -8.20947265625, "reward_std": 6.251974582672119, "rewards/rm_reward_func/mean": -8.20947265625, "rewards/rm_reward_func/std": 6.31874942779541, "step": 27 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 394.0, "completions/max_terminated_length": 394.0, "completions/mean_length": 167.78125, "completions/mean_terminated_length": 167.78125, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "epoch": 0.0224, "grad_norm": 4.310393333435059, "kl": 0.0003063678741455078, "learning_rate": 4.285714285714285e-07, "loss": 0.0078, "num_tokens": 350994.0, "reward": -4.7073516845703125, "reward_std": 3.5501272678375244, "rewards/rm_reward_func/mean": -4.7073516845703125, "rewards/rm_reward_func/std": 11.039022445678711, "step": 28 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 493.0, "completions/mean_length": 316.75, "completions/mean_terminated_length": 199.60000610351562, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "epoch": 0.0232, "grad_norm": 2.857746124267578, "kl": 0.0002956390380859375, "learning_rate": 4.444444444444444e-07, "loss": 0.0622, "num_tokens": 368186.0, "reward": -8.3876953125, "reward_std": 4.28533935546875, "rewards/rm_reward_func/mean": -8.3876953125, "rewards/rm_reward_func/std": 6.924084663391113, "step": 29 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 453.0, "completions/mean_length": 275.71875, "completions/mean_terminated_length": 221.19232177734375, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "epoch": 0.024, "grad_norm": 2.670747995376587, "kl": 0.0003662109375, "learning_rate": 4.6031746031746025e-07, "loss": 0.0619, "num_tokens": 379457.0, "reward": -8.928466796875, "reward_std": 8.38475227355957, "rewards/rm_reward_func/mean": -8.928466796875, "rewards/rm_reward_func/std": 13.34780502319336, "step": 30 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40625, "completions/max_length": 512.0, "completions/max_terminated_length": 470.0, "completions/mean_length": 298.5, "completions/mean_terminated_length": 152.42105102539062, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "epoch": 0.0248, "grad_norm": 2.7848010063171387, "kl": 0.00027370452880859375, "learning_rate": 4.761904761904761e-07, "loss": 0.1278, "num_tokens": 392985.0, "reward": -14.843505859375, "reward_std": 5.917230129241943, "rewards/rm_reward_func/mean": -14.843505859375, "rewards/rm_reward_func/std": 8.529240608215332, "step": 31 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 258.9375, "completions/mean_terminated_length": 174.58334350585938, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "epoch": 0.0256, "grad_norm": 4.443420886993408, "kl": 0.0003578662872314453, "learning_rate": 4.92063492063492e-07, "loss": 0.128, "num_tokens": 405471.0, "reward": -4.92132568359375, "reward_std": 4.653083801269531, "rewards/rm_reward_func/mean": -4.92132568359375, "rewards/rm_reward_func/std": 13.973512649536133, "step": 32 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 463.0, "completions/mean_length": 181.9375, "completions/mean_terminated_length": 159.93333435058594, "completions/min_length": 19.0, "completions/min_terminated_length": 19.0, "epoch": 0.0264, "grad_norm": 4.213375091552734, "kl": 0.0003600120544433594, "learning_rate": 5.079365079365079e-07, "loss": -0.1714, "num_tokens": 415629.0, "reward": -11.66253662109375, "reward_std": 6.659981727600098, "rewards/rm_reward_func/mean": -11.66253662109375, "rewards/rm_reward_func/std": 6.839272975921631, "step": 33 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 435.0, "completions/mean_length": 236.0625, "completions/mean_terminated_length": 158.8000030517578, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "epoch": 0.0272, "grad_norm": 3.4807815551757812, "kl": 0.00034689903259277344, "learning_rate": 5.238095238095238e-07, "loss": -0.0034, "num_tokens": 429039.0, "reward": -8.47998046875, "reward_std": 5.434866905212402, "rewards/rm_reward_func/mean": -8.47998046875, "rewards/rm_reward_func/std": 7.376414775848389, "step": 34 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 493.0, "completions/mean_length": 336.46875, "completions/mean_terminated_length": 244.52381896972656, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "epoch": 0.028, "grad_norm": 2.5712099075317383, "kl": 0.0003306865692138672, "learning_rate": 5.396825396825396e-07, "loss": 0.1245, "num_tokens": 443430.0, "reward": -7.155723571777344, "reward_std": 6.052016735076904, "rewards/rm_reward_func/mean": -7.155723571777344, "rewards/rm_reward_func/std": 8.14072322845459, "step": 35 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40625, "completions/max_length": 512.0, "completions/max_terminated_length": 230.0, "completions/mean_length": 251.21875, "completions/mean_terminated_length": 72.78947448730469, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "epoch": 0.0288, "grad_norm": 3.459679126739502, "kl": 0.0003533363342285156, "learning_rate": 5.555555555555555e-07, "loss": 0.027, "num_tokens": 454285.0, "reward": -9.2822265625, "reward_std": 5.530088424682617, "rewards/rm_reward_func/mean": -9.2822265625, "rewards/rm_reward_func/std": 6.309089183807373, "step": 36 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 426.0, "completions/mean_length": 222.09375, "completions/mean_terminated_length": 212.74192810058594, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "epoch": 0.0296, "grad_norm": 2.861426830291748, "kl": 0.0002856254577636719, "learning_rate": 5.714285714285714e-07, "loss": -0.0668, "num_tokens": 463344.0, "reward": -7.4073486328125, "reward_std": 6.660886287689209, "rewards/rm_reward_func/mean": -7.4073486328125, "rewards/rm_reward_func/std": 6.9068474769592285, "step": 37 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 188.90625, "completions/mean_terminated_length": 98.43999481201172, "completions/min_length": 12.0, "completions/min_terminated_length": 12.0, "epoch": 0.0304, "grad_norm": 5.866414546966553, "kl": 0.0003485679626464844, "learning_rate": 5.873015873015873e-07, "loss": -0.2332, "num_tokens": 475037.0, "reward": -9.38623046875, "reward_std": 3.1565802097320557, "rewards/rm_reward_func/mean": -9.38623046875, "rewards/rm_reward_func/std": 10.67844295501709, "step": 38 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 404.0, "completions/mean_length": 303.28125, "completions/mean_terminated_length": 221.60870361328125, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.0312, "grad_norm": 2.542703151702881, "kl": 0.00039768218994140625, "learning_rate": 6.031746031746031e-07, "loss": 0.076, "num_tokens": 487278.0, "reward": -8.93896484375, "reward_std": 3.1694929599761963, "rewards/rm_reward_func/mean": -8.93896484375, "rewards/rm_reward_func/std": 8.222332000732422, "step": 39 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 229.0625, "completions/mean_terminated_length": 163.7692413330078, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "epoch": 0.032, "grad_norm": 6.272686004638672, "kl": 0.0004992485046386719, "learning_rate": 6.19047619047619e-07, "loss": -0.0515, "num_tokens": 498952.0, "reward": -3.323486328125, "reward_std": 5.879536151885986, "rewards/rm_reward_func/mean": -3.323486328125, "rewards/rm_reward_func/std": 8.639906883239746, "step": 40 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 375.125, "completions/mean_terminated_length": 321.5652160644531, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "epoch": 0.0328, "grad_norm": 2.0339434146881104, "kl": 0.00039386749267578125, "learning_rate": 6.349206349206349e-07, "loss": -0.1041, "num_tokens": 513076.0, "reward": -4.98583984375, "reward_std": 4.426126956939697, "rewards/rm_reward_func/mean": -4.98583984375, "rewards/rm_reward_func/std": 5.537177085876465, "step": 41 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 397.0, "completions/mean_length": 161.84375, "completions/mean_terminated_length": 150.5483856201172, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "epoch": 0.0336, "grad_norm": 3.7768754959106445, "kl": 0.0003666877746582031, "learning_rate": 6.507936507936507e-07, "loss": -0.0783, "num_tokens": 523351.0, "reward": -8.8109130859375, "reward_std": 5.987880706787109, "rewards/rm_reward_func/mean": -8.8109130859375, "rewards/rm_reward_func/std": 6.827143669128418, "step": 42 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 206.0, "completions/mean_length": 207.6875, "completions/mean_terminated_length": 106.25, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "epoch": 0.0344, "grad_norm": 4.852839469909668, "kl": 0.0005655288696289062, "learning_rate": 6.666666666666666e-07, "loss": 0.0577, "num_tokens": 532325.0, "reward": -11.9395751953125, "reward_std": 3.7294559478759766, "rewards/rm_reward_func/mean": -11.9395751953125, "rewards/rm_reward_func/std": 6.000690460205078, "step": 43 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 493.0, "completions/mean_length": 345.21875, "completions/mean_terminated_length": 339.8387145996094, "completions/min_length": 116.0, "completions/min_terminated_length": 116.0, "epoch": 0.0352, "grad_norm": 2.18628191947937, "kl": 0.0003714561462402344, "learning_rate": 6.825396825396826e-07, "loss": -0.0234, "num_tokens": 546108.0, "reward": -8.53497314453125, "reward_std": 4.388243675231934, "rewards/rm_reward_func/mean": -8.53497314453125, "rewards/rm_reward_func/std": 7.794600486755371, "step": 44 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 388.0, "completions/mean_length": 305.90625, "completions/mean_terminated_length": 197.95237731933594, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "epoch": 0.036, "grad_norm": 2.3126137256622314, "kl": 0.0004107952117919922, "learning_rate": 6.984126984126983e-07, "loss": 0.0486, "num_tokens": 559873.0, "reward": -7.420013427734375, "reward_std": 7.450569152832031, "rewards/rm_reward_func/mean": -7.420013427734375, "rewards/rm_reward_func/std": 10.354388236999512, "step": 45 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 387.25, "completions/mean_terminated_length": 321.9047546386719, "completions/min_length": 122.0, "completions/min_terminated_length": 122.0, "epoch": 0.0368, "grad_norm": 2.344557762145996, "kl": 0.00038242340087890625, "learning_rate": 7.142857142857143e-07, "loss": -0.011, "num_tokens": 575289.0, "reward": -4.837789535522461, "reward_std": 7.047362327575684, "rewards/rm_reward_func/mean": -4.837789535522461, "rewards/rm_reward_func/std": 11.340645790100098, "step": 46 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 233.0, "completions/mean_length": 140.09375, "completions/mean_terminated_length": 71.22222137451172, "completions/min_length": 20.0, "completions/min_terminated_length": 20.0, "epoch": 0.0376, "grad_norm": 5.920055389404297, "kl": 0.0005631446838378906, "learning_rate": 7.301587301587301e-07, "loss": -0.1061, "num_tokens": 586604.0, "reward": -4.19140625, "reward_std": 4.153356552124023, "rewards/rm_reward_func/mean": -4.19140625, "rewards/rm_reward_func/std": 7.545008182525635, "step": 47 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 378.0, "completions/max_terminated_length": 378.0, "completions/mean_length": 163.6875, "completions/mean_terminated_length": 163.6875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "epoch": 0.0384, "grad_norm": 3.746752977371216, "kl": 0.0006389617919921875, "learning_rate": 7.46031746031746e-07, "loss": -0.094, "num_tokens": 593642.0, "reward": -10.421875, "reward_std": 5.012988567352295, "rewards/rm_reward_func/mean": -10.421875, "rewards/rm_reward_func/std": 6.197994232177734, "step": 48 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 333.0625, "completions/mean_terminated_length": 251.72727966308594, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "epoch": 0.0392, "grad_norm": 3.094766139984131, "kl": 0.0006709098815917969, "learning_rate": 7.619047619047618e-07, "loss": 0.0426, "num_tokens": 606588.0, "reward": -7.983154296875, "reward_std": 8.31886100769043, "rewards/rm_reward_func/mean": -7.983154296875, "rewards/rm_reward_func/std": 10.532109260559082, "step": 49 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 229.5, "completions/mean_terminated_length": 200.27586364746094, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "epoch": 0.04, "grad_norm": 3.4759812355041504, "kl": 0.0006818771362304688, "learning_rate": 7.777777777777778e-07, "loss": 0.1175, "num_tokens": 616556.0, "reward": -6.1270751953125, "reward_std": 8.114751815795898, "rewards/rm_reward_func/mean": -6.1270751953125, "rewards/rm_reward_func/std": 8.840689659118652, "step": 50 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 397.0, "completions/mean_length": 311.0625, "completions/mean_terminated_length": 205.8095245361328, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "epoch": 0.0408, "grad_norm": 2.890885353088379, "kl": 0.000766754150390625, "learning_rate": 7.936507936507936e-07, "loss": 0.0924, "num_tokens": 628694.0, "reward": -8.6982421875, "reward_std": 3.657209873199463, "rewards/rm_reward_func/mean": -8.6982421875, "rewards/rm_reward_func/std": 13.067253112792969, "step": 51 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 283.28125, "completions/mean_terminated_length": 163.4761962890625, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "epoch": 0.0416, "grad_norm": 2.9035823345184326, "kl": 0.0004100799560546875, "learning_rate": 8.095238095238095e-07, "loss": 0.2123, "num_tokens": 641223.0, "reward": -5.83807373046875, "reward_std": 4.900543689727783, "rewards/rm_reward_func/mean": -5.83807373046875, "rewards/rm_reward_func/std": 7.653470993041992, "step": 52 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 425.0, "completions/max_terminated_length": 425.0, "completions/mean_length": 114.9375, "completions/mean_terminated_length": 114.9375, "completions/min_length": 6.0, "completions/min_terminated_length": 6.0, "epoch": 0.0424, "grad_norm": 10.671420097351074, "kl": 0.0011627674102783203, "learning_rate": 8.253968253968253e-07, "loss": -0.1026, "num_tokens": 647501.0, "reward": -8.42413330078125, "reward_std": 5.442896842956543, "rewards/rm_reward_func/mean": -8.42413330078125, "rewards/rm_reward_func/std": 8.57506275177002, "step": 53 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 468.0, "completions/mean_length": 252.53125, "completions/mean_terminated_length": 179.87998962402344, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "epoch": 0.0432, "grad_norm": 3.140366315841675, "kl": 0.0007600784301757812, "learning_rate": 8.412698412698413e-07, "loss": 0.1312, "num_tokens": 660070.0, "reward": -10.078125, "reward_std": 4.523303985595703, "rewards/rm_reward_func/mean": -10.078125, "rewards/rm_reward_func/std": 6.438615322113037, "step": 54 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 202.15625, "completions/mean_terminated_length": 157.8928680419922, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "epoch": 0.044, "grad_norm": 4.555398941040039, "kl": 0.0013036727905273438, "learning_rate": 8.57142857142857e-07, "loss": 0.0684, "num_tokens": 669715.0, "reward": -10.841552734375, "reward_std": 8.45435619354248, "rewards/rm_reward_func/mean": -10.841552734375, "rewards/rm_reward_func/std": 9.872593879699707, "step": 55 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.4375, "completions/max_length": 512.0, "completions/max_terminated_length": 479.0, "completions/mean_length": 336.65625, "completions/mean_terminated_length": 200.2777862548828, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "epoch": 0.0448, "grad_norm": 2.7749080657958984, "kl": 0.0010623931884765625, "learning_rate": 8.73015873015873e-07, "loss": -0.0338, "num_tokens": 683152.0, "reward": -1.090301513671875, "reward_std": 3.921980142593384, "rewards/rm_reward_func/mean": -1.090301513671875, "rewards/rm_reward_func/std": 6.629665374755859, "step": 56 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 320.03125, "completions/mean_terminated_length": 244.91305541992188, "completions/min_length": 20.0, "completions/min_terminated_length": 20.0, "epoch": 0.0456, "grad_norm": 2.539121389389038, "kl": 0.0009260177612304688, "learning_rate": 8.888888888888888e-07, "loss": 0.1425, "num_tokens": 697105.0, "reward": -7.415740966796875, "reward_std": 5.969917297363281, "rewards/rm_reward_func/mean": -7.415740966796875, "rewards/rm_reward_func/std": 7.336832523345947, "step": 57 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40625, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 330.71875, "completions/mean_terminated_length": 206.68421936035156, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "epoch": 0.0464, "grad_norm": 2.4101502895355225, "kl": 0.0011167526245117188, "learning_rate": 9.047619047619047e-07, "loss": -0.0899, "num_tokens": 710128.0, "reward": -6.55548095703125, "reward_std": 5.115406036376953, "rewards/rm_reward_func/mean": -6.55548095703125, "rewards/rm_reward_func/std": 6.5659332275390625, "step": 58 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 307.375, "completions/mean_terminated_length": 239.1666717529297, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "epoch": 0.0472, "grad_norm": 2.872269630432129, "kl": 0.0010232925415039062, "learning_rate": 9.206349206349205e-07, "loss": 0.014, "num_tokens": 721980.0, "reward": -5.94189453125, "reward_std": 5.278933525085449, "rewards/rm_reward_func/mean": -5.94189453125, "rewards/rm_reward_func/std": 8.825187683105469, "step": 59 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 475.0, "completions/mean_length": 270.4375, "completions/mean_terminated_length": 235.9285888671875, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "epoch": 0.048, "grad_norm": 3.2065460681915283, "kl": 0.0014667510986328125, "learning_rate": 9.365079365079365e-07, "loss": 0.2253, "num_tokens": 733306.0, "reward": -5.34954833984375, "reward_std": 8.191361427307129, "rewards/rm_reward_func/mean": -5.34954833984375, "rewards/rm_reward_func/std": 10.043052673339844, "step": 60 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 470.0, "completions/mean_length": 302.96875, "completions/mean_terminated_length": 254.73077392578125, "completions/min_length": 10.0, "completions/min_terminated_length": 10.0, "epoch": 0.0488, "grad_norm": 2.4722800254821777, "kl": 0.0020198822021484375, "learning_rate": 9.523809523809522e-07, "loss": -0.0115, "num_tokens": 745889.0, "reward": -13.30328369140625, "reward_std": 3.812379837036133, "rewards/rm_reward_func/mean": -13.30328369140625, "rewards/rm_reward_func/std": 8.161885261535645, "step": 61 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 218.5625, "completions/mean_terminated_length": 209.09677124023438, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "epoch": 0.0496, "grad_norm": 3.010167360305786, "kl": 0.0018796920776367188, "learning_rate": 9.682539682539682e-07, "loss": -0.2943, "num_tokens": 754907.0, "reward": -7.835418701171875, "reward_std": 5.403800010681152, "rewards/rm_reward_func/mean": -7.835418701171875, "rewards/rm_reward_func/std": 8.523995399475098, "step": 62 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 455.0, "completions/mean_length": 331.15625, "completions/mean_terminated_length": 248.95455932617188, "completions/min_length": 92.0, "completions/min_terminated_length": 92.0, "epoch": 0.0504, "grad_norm": 2.819601058959961, "kl": 0.0015172958374023438, "learning_rate": 9.84126984126984e-07, "loss": -0.0081, "num_tokens": 768560.0, "reward": -11.4364013671875, "reward_std": 3.361478328704834, "rewards/rm_reward_func/mean": -11.4364013671875, "rewards/rm_reward_func/std": 6.750278472900391, "step": 63 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 485.0, "completions/mean_length": 327.03125, "completions/mean_terminated_length": 230.1428680419922, "completions/min_length": 16.0, "completions/min_terminated_length": 16.0, "epoch": 0.0512, "grad_norm": 3.7442524433135986, "kl": 0.0022602081298828125, "learning_rate": 1e-06, "loss": -0.1989, "num_tokens": 781345.0, "reward": -8.39996337890625, "reward_std": 4.888169288635254, "rewards/rm_reward_func/mean": -8.39996337890625, "rewards/rm_reward_func/std": 5.147318363189697, "step": 64 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 268.40625, "completions/mean_terminated_length": 252.16668701171875, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "epoch": 0.052, "grad_norm": 2.8628954887390137, "kl": 0.0019245147705078125, "learning_rate": 1e-06, "loss": -0.0962, "num_tokens": 792750.0, "reward": -7.360107421875, "reward_std": 5.436917781829834, "rewards/rm_reward_func/mean": -7.360107421875, "rewards/rm_reward_func/std": 11.22769832611084, "step": 65 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40625, "completions/max_length": 512.0, "completions/max_terminated_length": 493.0, "completions/mean_length": 332.46875, "completions/mean_terminated_length": 209.63157653808594, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "epoch": 0.0528, "grad_norm": 3.654360294342041, "kl": 0.0015087127685546875, "learning_rate": 1e-06, "loss": -0.1237, "num_tokens": 805853.0, "reward": -9.6962890625, "reward_std": 2.946159839630127, "rewards/rm_reward_func/mean": -9.6962890625, "rewards/rm_reward_func/std": 4.295775890350342, "step": 66 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.53125, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 402.0, "completions/mean_terminated_length": 277.3333435058594, "completions/min_length": 82.0, "completions/min_terminated_length": 82.0, "epoch": 0.0536, "grad_norm": 2.3473589420318604, "kl": 0.002231597900390625, "learning_rate": 1e-06, "loss": -0.0437, "num_tokens": 823173.0, "reward": -8.37060546875, "reward_std": 3.5887060165405273, "rewards/rm_reward_func/mean": -8.37060546875, "rewards/rm_reward_func/std": 4.726972579956055, "step": 67 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.4375, "completions/max_length": 512.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 364.6875, "completions/mean_terminated_length": 250.11111450195312, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "epoch": 0.0544, "grad_norm": 2.694181442260742, "kl": 0.001918792724609375, "learning_rate": 1e-06, "loss": 0.2082, "num_tokens": 837739.0, "reward": -7.8717041015625, "reward_std": 3.8687283992767334, "rewards/rm_reward_func/mean": -7.8717041015625, "rewards/rm_reward_func/std": 7.253809928894043, "step": 68 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 312.46875, "completions/mean_terminated_length": 256.6000061035156, "completions/min_length": 85.0, "completions/min_terminated_length": 85.0, "epoch": 0.0552, "grad_norm": 2.733241558074951, "kl": 0.002780914306640625, "learning_rate": 1e-06, "loss": -0.0181, "num_tokens": 849434.0, "reward": -6.937103271484375, "reward_std": 3.667171001434326, "rewards/rm_reward_func/mean": -6.937103271484375, "rewards/rm_reward_func/std": 5.047467231750488, "step": 69 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 336.71875, "completions/mean_terminated_length": 278.29168701171875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "epoch": 0.056, "grad_norm": 2.4156734943389893, "kl": 0.002368927001953125, "learning_rate": 1e-06, "loss": -0.1443, "num_tokens": 862321.0, "reward": -4.07220458984375, "reward_std": 10.273677825927734, "rewards/rm_reward_func/mean": -4.07220458984375, "rewards/rm_reward_func/std": 13.013540267944336, "step": 70 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40625, "completions/max_length": 512.0, "completions/max_terminated_length": 419.0, "completions/mean_length": 314.90625, "completions/mean_terminated_length": 180.05262756347656, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "epoch": 0.0568, "grad_norm": 3.1217379570007324, "kl": 0.00289154052734375, "learning_rate": 1e-06, "loss": -0.0551, "num_tokens": 875374.0, "reward": -6.78515625, "reward_std": 4.401189804077148, "rewards/rm_reward_func/mean": -6.78515625, "rewards/rm_reward_func/std": 7.016132354736328, "step": 71 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 353.125, "completions/mean_terminated_length": 323.7037048339844, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "epoch": 0.0576, "grad_norm": 2.9577646255493164, "kl": 0.003635406494140625, "learning_rate": 1e-06, "loss": -0.1052, "num_tokens": 889138.0, "reward": -3.15277099609375, "reward_std": 5.309854507446289, "rewards/rm_reward_func/mean": -3.15277099609375, "rewards/rm_reward_func/std": 5.763915538787842, "step": 72 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 466.0, "completions/mean_length": 258.90625, "completions/mean_terminated_length": 188.0399932861328, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "epoch": 0.0584, "grad_norm": 3.2543458938598633, "kl": 0.0028629302978515625, "learning_rate": 1e-06, "loss": 0.2214, "num_tokens": 900951.0, "reward": -8.47222900390625, "reward_std": 4.21871280670166, "rewards/rm_reward_func/mean": -8.47222900390625, "rewards/rm_reward_func/std": 5.818933486938477, "step": 73 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 414.0, "completions/mean_length": 243.0, "completions/mean_terminated_length": 153.33334350585938, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "epoch": 0.0592, "grad_norm": 3.68485426902771, "kl": 0.0032253265380859375, "learning_rate": 1e-06, "loss": -0.0463, "num_tokens": 911351.0, "reward": -9.206695556640625, "reward_std": 6.0724778175354, "rewards/rm_reward_func/mean": -9.206695556640625, "rewards/rm_reward_func/std": 7.073054313659668, "step": 74 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 472.0, "completions/mean_length": 289.78125, "completions/mean_terminated_length": 238.50001525878906, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "epoch": 0.06, "grad_norm": 2.78497052192688, "kl": 0.00466156005859375, "learning_rate": 1e-06, "loss": 0.2228, "num_tokens": 923680.0, "reward": -6.83343505859375, "reward_std": 7.71270227432251, "rewards/rm_reward_func/mean": -6.83343505859375, "rewards/rm_reward_func/std": 10.820587158203125, "step": 75 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.71875, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 481.46875, "completions/mean_terminated_length": 403.4444580078125, "completions/min_length": 134.0, "completions/min_terminated_length": 134.0, "epoch": 0.0608, "grad_norm": 1.8886445760726929, "kl": 0.002117633819580078, "learning_rate": 1e-06, "loss": 0.0654, "num_tokens": 944191.0, "reward": -10.4248046875, "reward_std": 5.140851020812988, "rewards/rm_reward_func/mean": -10.4248046875, "rewards/rm_reward_func/std": 9.057696342468262, "step": 76 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 459.0, "completions/mean_length": 180.1875, "completions/mean_terminated_length": 169.48387145996094, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "epoch": 0.0616, "grad_norm": 5.139257431030273, "kl": 0.002521514892578125, "learning_rate": 1e-06, "loss": 0.1781, "num_tokens": 952477.0, "reward": -1.8204498291015625, "reward_std": 4.737975120544434, "rewards/rm_reward_func/mean": -1.8204498291015625, "rewards/rm_reward_func/std": 12.699488639831543, "step": 77 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 411.0, "completions/mean_length": 212.5625, "completions/mean_terminated_length": 181.58621215820312, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "epoch": 0.0624, "grad_norm": 3.0258443355560303, "kl": 0.0031642913818359375, "learning_rate": 1e-06, "loss": 0.2051, "num_tokens": 963415.0, "reward": -5.3250732421875, "reward_std": 4.80600643157959, "rewards/rm_reward_func/mean": -5.3250732421875, "rewards/rm_reward_func/std": 4.75626802444458, "step": 78 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 343.53125, "completions/mean_terminated_length": 277.60870361328125, "completions/min_length": 40.0, "completions/min_terminated_length": 40.0, "epoch": 0.0632, "grad_norm": 2.744492292404175, "kl": 0.00372314453125, "learning_rate": 1e-06, "loss": -0.0128, "num_tokens": 977392.0, "reward": -4.4423980712890625, "reward_std": 5.815633773803711, "rewards/rm_reward_func/mean": -4.4423980712890625, "rewards/rm_reward_func/std": 10.560279846191406, "step": 79 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.4375, "completions/max_length": 512.0, "completions/max_terminated_length": 466.0, "completions/mean_length": 312.25, "completions/mean_terminated_length": 156.88888549804688, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "epoch": 0.064, "grad_norm": 2.8983473777770996, "kl": 0.003662109375, "learning_rate": 1e-06, "loss": 0.042, "num_tokens": 995120.0, "reward": -12.873046875, "reward_std": 3.767939805984497, "rewards/rm_reward_func/mean": -12.873046875, "rewards/rm_reward_func/std": 7.584336757659912, "step": 80 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 413.0, "completions/mean_length": 246.96875, "completions/mean_terminated_length": 185.8076934814453, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "epoch": 0.0648, "grad_norm": 3.573094129562378, "kl": 0.0052947998046875, "learning_rate": 1e-06, "loss": -0.1861, "num_tokens": 1005943.0, "reward": -5.154052734375, "reward_std": 5.333109378814697, "rewards/rm_reward_func/mean": -5.154052734375, "rewards/rm_reward_func/std": 5.350836277008057, "step": 81 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 470.0, "completions/mean_length": 298.96875, "completions/mean_terminated_length": 249.80770874023438, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "epoch": 0.0656, "grad_norm": 2.644007444381714, "kl": 0.003917694091796875, "learning_rate": 1e-06, "loss": 0.0187, "num_tokens": 1020238.0, "reward": -9.076370239257812, "reward_std": 4.40856409072876, "rewards/rm_reward_func/mean": -9.076370239257812, "rewards/rm_reward_func/std": 6.288516521453857, "step": 82 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 465.0, "completions/mean_length": 307.0, "completions/mean_terminated_length": 199.61904907226562, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "epoch": 0.0664, "grad_norm": 2.8343067169189453, "kl": 0.0068359375, "learning_rate": 1e-06, "loss": -0.0099, "num_tokens": 1032262.0, "reward": -7.667877197265625, "reward_std": 5.629947185516357, "rewards/rm_reward_func/mean": -7.667877197265625, "rewards/rm_reward_func/std": 7.898858070373535, "step": 83 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 465.0, "completions/max_terminated_length": 465.0, "completions/mean_length": 225.71875, "completions/mean_terminated_length": 225.71875, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "epoch": 0.0672, "grad_norm": 2.8555967807769775, "kl": 0.00545501708984375, "learning_rate": 1e-06, "loss": 0.0421, "num_tokens": 1042485.0, "reward": 2.5863037109375, "reward_std": 3.6911566257476807, "rewards/rm_reward_func/mean": 2.5863037109375, "rewards/rm_reward_func/std": 10.012650489807129, "step": 84 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 435.0, "completions/mean_length": 219.59375, "completions/mean_terminated_length": 210.16128540039062, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "epoch": 0.068, "grad_norm": 4.213369369506836, "kl": 0.00635528564453125, "learning_rate": 1e-06, "loss": 0.0863, "num_tokens": 1052608.0, "reward": -3.3102264404296875, "reward_std": 3.797806978225708, "rewards/rm_reward_func/mean": -3.3102264404296875, "rewards/rm_reward_func/std": 5.886042594909668, "step": 85 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 289.65625, "completions/mean_terminated_length": 266.6551818847656, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "epoch": 0.0688, "grad_norm": 3.4353296756744385, "kl": 0.0052337646484375, "learning_rate": 1e-06, "loss": 0.0776, "num_tokens": 1064949.0, "reward": 2.492401123046875, "reward_std": 7.244277000427246, "rewards/rm_reward_func/mean": 2.492401123046875, "rewards/rm_reward_func/std": 9.47168254852295, "step": 86 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 236.28125, "completions/mean_terminated_length": 196.8928680419922, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "epoch": 0.0696, "grad_norm": 9.39013385772705, "kl": 0.00577545166015625, "learning_rate": 1e-06, "loss": -0.0612, "num_tokens": 1074950.0, "reward": -5.302734375, "reward_std": 6.743742942810059, "rewards/rm_reward_func/mean": -5.302734375, "rewards/rm_reward_func/std": 8.008939743041992, "step": 87 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 474.0, "completions/mean_length": 292.46875, "completions/mean_terminated_length": 192.68182373046875, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "epoch": 0.0704, "grad_norm": 3.4955945014953613, "kl": 0.0057942867279052734, "learning_rate": 1e-06, "loss": 0.2027, "num_tokens": 1090957.0, "reward": -4.89019775390625, "reward_std": 5.86878776550293, "rewards/rm_reward_func/mean": -4.89019775390625, "rewards/rm_reward_func/std": 11.281463623046875, "step": 88 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40625, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 299.4375, "completions/mean_terminated_length": 154.0, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "epoch": 0.0712, "grad_norm": 3.3145103454589844, "kl": 0.00629425048828125, "learning_rate": 1e-06, "loss": 0.0759, "num_tokens": 1103699.0, "reward": -7.9558868408203125, "reward_std": 3.8132681846618652, "rewards/rm_reward_func/mean": -7.9558868408203125, "rewards/rm_reward_func/std": 5.60645055770874, "step": 89 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 462.0, "completions/mean_length": 242.3125, "completions/mean_terminated_length": 214.41378784179688, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "epoch": 0.072, "grad_norm": 3.746366500854492, "kl": 0.007389068603515625, "learning_rate": 1e-06, "loss": 0.0806, "num_tokens": 1114293.0, "reward": 4.7412109375, "reward_std": 6.3503098487854, "rewards/rm_reward_func/mean": 4.7412109375, "rewards/rm_reward_func/std": 12.657453536987305, "step": 90 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.59375, "completions/max_length": 512.0, "completions/max_terminated_length": 463.0, "completions/mean_length": 392.34375, "completions/mean_terminated_length": 217.4615478515625, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "epoch": 0.0728, "grad_norm": 2.246333360671997, "kl": 0.00717926025390625, "learning_rate": 1e-06, "loss": -0.0118, "num_tokens": 1130400.0, "reward": -6.076568603515625, "reward_std": 4.733026027679443, "rewards/rm_reward_func/mean": -6.076568603515625, "rewards/rm_reward_func/std": 10.103605270385742, "step": 91 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 402.0, "completions/mean_length": 368.15625, "completions/mean_terminated_length": 292.8095397949219, "completions/min_length": 19.0, "completions/min_terminated_length": 19.0, "epoch": 0.0736, "grad_norm": 2.1302545070648193, "kl": 0.007269859313964844, "learning_rate": 1e-06, "loss": 0.1118, "num_tokens": 1146957.0, "reward": -7.0111083984375, "reward_std": 6.281430244445801, "rewards/rm_reward_func/mean": -7.0111083984375, "rewards/rm_reward_func/std": 10.263130187988281, "step": 92 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 339.0, "completions/mean_terminated_length": 271.3043518066406, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "epoch": 0.0744, "grad_norm": 3.688321113586426, "kl": 0.00998687744140625, "learning_rate": 1e-06, "loss": 0.0504, "num_tokens": 1159773.0, "reward": -3.8799972534179688, "reward_std": 4.736905097961426, "rewards/rm_reward_func/mean": -3.8799972534179688, "rewards/rm_reward_func/std": 6.819652557373047, "step": 93 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 484.0, "completions/mean_length": 260.21875, "completions/mean_terminated_length": 252.09677124023438, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "epoch": 0.0752, "grad_norm": 3.679842472076416, "kl": 0.00600433349609375, "learning_rate": 1e-06, "loss": -0.1287, "num_tokens": 1171236.0, "reward": -5.69622802734375, "reward_std": 3.642548084259033, "rewards/rm_reward_func/mean": -5.69622802734375, "rewards/rm_reward_func/std": 5.160977363586426, "step": 94 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40625, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 376.25, "completions/mean_terminated_length": 283.368408203125, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "epoch": 0.076, "grad_norm": 2.582540273666382, "kl": 0.00710296630859375, "learning_rate": 1e-06, "loss": -0.1972, "num_tokens": 1186820.0, "reward": -8.5262451171875, "reward_std": 3.4011950492858887, "rewards/rm_reward_func/mean": -8.5262451171875, "rewards/rm_reward_func/std": 7.117488384246826, "step": 95 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 423.0, "completions/mean_length": 178.6875, "completions/mean_terminated_length": 167.93548583984375, "completions/min_length": 23.0, "completions/min_terminated_length": 23.0, "epoch": 0.0768, "grad_norm": 4.087407112121582, "kl": 0.0051116943359375, "learning_rate": 1e-06, "loss": -0.1202, "num_tokens": 1198074.0, "reward": -5.597259521484375, "reward_std": 7.2116804122924805, "rewards/rm_reward_func/mean": -5.597259521484375, "rewards/rm_reward_func/std": 8.06857681274414, "step": 96 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 284.03125, "completions/mean_terminated_length": 208.0416717529297, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "epoch": 0.0776, "grad_norm": 9.104924201965332, "kl": 0.00815582275390625, "learning_rate": 1e-06, "loss": -0.1018, "num_tokens": 1211131.0, "reward": -9.880126953125, "reward_std": 6.673112869262695, "rewards/rm_reward_func/mean": -9.880126953125, "rewards/rm_reward_func/std": 7.890270233154297, "step": 97 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 318.0, "completions/mean_length": 260.78125, "completions/mean_terminated_length": 177.0416717529297, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "epoch": 0.0784, "grad_norm": 5.061870098114014, "kl": 0.0068988800048828125, "learning_rate": 1e-06, "loss": 0.1266, "num_tokens": 1224396.0, "reward": -9.5322265625, "reward_std": 5.915492057800293, "rewards/rm_reward_func/mean": -9.5322265625, "rewards/rm_reward_func/std": 13.454625129699707, "step": 98 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 312.5, "completions/mean_terminated_length": 275.5555725097656, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "epoch": 0.0792, "grad_norm": 7.630608081817627, "kl": 0.008739471435546875, "learning_rate": 1e-06, "loss": -0.0647, "num_tokens": 1239732.0, "reward": 1.023895263671875, "reward_std": 8.057183265686035, "rewards/rm_reward_func/mean": 1.023895263671875, "rewards/rm_reward_func/std": 10.845428466796875, "step": 99 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 415.0, "completions/max_terminated_length": 415.0, "completions/mean_length": 252.8125, "completions/mean_terminated_length": 252.8125, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "epoch": 0.08, "grad_norm": 3.1217799186706543, "kl": 0.01123046875, "learning_rate": 1e-06, "loss": -0.0374, "num_tokens": 1250262.0, "reward": -8.40460205078125, "reward_std": 3.330627679824829, "rewards/rm_reward_func/mean": -8.40460205078125, "rewards/rm_reward_func/std": 4.349534034729004, "step": 100 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 329.0, "completions/mean_length": 245.09375, "completions/mean_terminated_length": 123.7727279663086, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "epoch": 0.0808, "grad_norm": 4.933348655700684, "kl": 0.0134124755859375, "learning_rate": 1e-06, "loss": 0.1239, "num_tokens": 1260473.0, "reward": -4.2023468017578125, "reward_std": 4.068781852722168, "rewards/rm_reward_func/mean": -4.2023468017578125, "rewards/rm_reward_func/std": 4.7899885177612305, "step": 101 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 316.03125, "completions/mean_terminated_length": 270.8077087402344, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "epoch": 0.0816, "grad_norm": 5.183587551116943, "kl": 0.01630401611328125, "learning_rate": 1e-06, "loss": -0.0734, "num_tokens": 1272762.0, "reward": -2.0743408203125, "reward_std": 4.2065324783325195, "rewards/rm_reward_func/mean": -2.0743408203125, "rewards/rm_reward_func/std": 7.418477535247803, "step": 102 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 190.34375, "completions/mean_terminated_length": 144.3928680419922, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "epoch": 0.0824, "grad_norm": 9.615796089172363, "kl": 0.0133819580078125, "learning_rate": 1e-06, "loss": 0.2496, "num_tokens": 1282085.0, "reward": -2.1240234375, "reward_std": 6.015683650970459, "rewards/rm_reward_func/mean": -2.1240234375, "rewards/rm_reward_func/std": 7.95778751373291, "step": 103 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.4375, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 318.75, "completions/mean_terminated_length": 168.44444274902344, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "epoch": 0.0832, "grad_norm": 4.339044094085693, "kl": 0.01284027099609375, "learning_rate": 1e-06, "loss": 0.0598, "num_tokens": 1298069.0, "reward": -1.1455078125, "reward_std": 5.1649861335754395, "rewards/rm_reward_func/mean": -1.1455078125, "rewards/rm_reward_func/std": 7.434368133544922, "step": 104 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.65625, "completions/max_length": 512.0, "completions/max_terminated_length": 491.0, "completions/mean_length": 463.78125, "completions/mean_terminated_length": 371.727294921875, "completions/min_length": 202.0, "completions/min_terminated_length": 202.0, "epoch": 0.084, "grad_norm": 3.03131365776062, "kl": 0.0128326416015625, "learning_rate": 1e-06, "loss": -0.0067, "num_tokens": 1315710.0, "reward": -4.12103271484375, "reward_std": 5.871215343475342, "rewards/rm_reward_func/mean": -4.12103271484375, "rewards/rm_reward_func/std": 8.927309036254883, "step": 105 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40625, "completions/max_length": 512.0, "completions/max_terminated_length": 449.0, "completions/mean_length": 384.53125, "completions/mean_terminated_length": 297.3157958984375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "epoch": 0.0848, "grad_norm": 2.725952625274658, "kl": 0.010955810546875, "learning_rate": 1e-06, "loss": 0.0927, "num_tokens": 1331855.0, "reward": -1.3194122314453125, "reward_std": 3.2353854179382324, "rewards/rm_reward_func/mean": -1.3194122314453125, "rewards/rm_reward_func/std": 4.594810485839844, "step": 106 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 463.0, "completions/mean_length": 315.5625, "completions/mean_terminated_length": 250.08334350585938, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "epoch": 0.0856, "grad_norm": 29.571395874023438, "kl": 0.010345458984375, "learning_rate": 1e-06, "loss": -0.1343, "num_tokens": 1345001.0, "reward": -6.830474853515625, "reward_std": 4.087754726409912, "rewards/rm_reward_func/mean": -6.830474853515625, "rewards/rm_reward_func/std": 9.476208686828613, "step": 107 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 291.5625, "completions/mean_terminated_length": 240.69232177734375, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "epoch": 0.0864, "grad_norm": 4.20673131942749, "kl": 0.013458251953125, "learning_rate": 1e-06, "loss": 0.1504, "num_tokens": 1357331.0, "reward": -5.711700439453125, "reward_std": 3.702240467071533, "rewards/rm_reward_func/mean": -5.711700439453125, "rewards/rm_reward_func/std": 7.0592451095581055, "step": 108 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 364.8125, "completions/mean_terminated_length": 297.9090881347656, "completions/min_length": 112.0, "completions/min_terminated_length": 112.0, "epoch": 0.0872, "grad_norm": 2.5211193561553955, "kl": 0.0170440673828125, "learning_rate": 1e-06, "loss": -0.0797, "num_tokens": 1371525.0, "reward": 3.977081298828125, "reward_std": 5.856342792510986, "rewards/rm_reward_func/mean": 3.977081298828125, "rewards/rm_reward_func/std": 12.513916969299316, "step": 109 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 464.0, "completions/mean_length": 319.65625, "completions/mean_terminated_length": 255.5416717529297, "completions/min_length": 122.0, "completions/min_terminated_length": 122.0, "epoch": 0.088, "grad_norm": 4.921560287475586, "kl": 0.01397705078125, "learning_rate": 1e-06, "loss": 0.0172, "num_tokens": 1384498.0, "reward": -5.39752197265625, "reward_std": 3.883021354675293, "rewards/rm_reward_func/mean": -5.39752197265625, "rewards/rm_reward_func/std": 11.68447208404541, "step": 110 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 317.4375, "completions/mean_terminated_length": 215.52381896972656, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "epoch": 0.0888, "grad_norm": 5.259191989898682, "kl": 0.0244293212890625, "learning_rate": 1e-06, "loss": 0.0037, "num_tokens": 1398792.0, "reward": -9.126190185546875, "reward_std": 3.321913719177246, "rewards/rm_reward_func/mean": -9.126190185546875, "rewards/rm_reward_func/std": 8.484477043151855, "step": 111 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 512.0, "completions/max_terminated_length": 411.0, "completions/mean_length": 360.4375, "completions/mean_terminated_length": 208.875, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "epoch": 0.0896, "grad_norm": 13.804244995117188, "kl": 0.0152130126953125, "learning_rate": 1e-06, "loss": 0.132, "num_tokens": 1413758.0, "reward": -12.7744140625, "reward_std": 3.5870838165283203, "rewards/rm_reward_func/mean": -12.7744140625, "rewards/rm_reward_func/std": 4.53514289855957, "step": 112 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.4375, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 399.8125, "completions/mean_terminated_length": 312.5555725097656, "completions/min_length": 150.0, "completions/min_terminated_length": 150.0, "epoch": 0.0904, "grad_norm": 2.4952642917633057, "kl": 0.00986480712890625, "learning_rate": 1e-06, "loss": -0.0065, "num_tokens": 1430304.0, "reward": -8.034423828125, "reward_std": 2.978208541870117, "rewards/rm_reward_func/mean": -8.034423828125, "rewards/rm_reward_func/std": 7.964836120605469, "step": 113 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.625, "completions/max_length": 512.0, "completions/max_terminated_length": 454.0, "completions/mean_length": 431.5, "completions/mean_terminated_length": 297.3333435058594, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "epoch": 0.0912, "grad_norm": 3.2031984329223633, "kl": 0.012176513671875, "learning_rate": 1e-06, "loss": -0.0707, "num_tokens": 1446624.0, "reward": -8.611083984375, "reward_std": 3.2820539474487305, "rewards/rm_reward_func/mean": -8.611083984375, "rewards/rm_reward_func/std": 6.850192070007324, "step": 114 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 437.0, "completions/mean_length": 244.6875, "completions/mean_terminated_length": 104.66667175292969, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "epoch": 0.092, "grad_norm": 17.274948120117188, "kl": 0.0135955810546875, "learning_rate": 1e-06, "loss": 0.3859, "num_tokens": 1461166.0, "reward": -5.8297882080078125, "reward_std": 4.278718948364258, "rewards/rm_reward_func/mean": -5.8297882080078125, "rewards/rm_reward_func/std": 5.697154521942139, "step": 115 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 486.0, "completions/mean_length": 312.5, "completions/mean_terminated_length": 221.8181915283203, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "epoch": 0.0928, "grad_norm": 3.2622101306915283, "kl": 0.01324462890625, "learning_rate": 1e-06, "loss": -0.178, "num_tokens": 1474222.0, "reward": -2.70849609375, "reward_std": 5.48113489151001, "rewards/rm_reward_func/mean": -2.70849609375, "rewards/rm_reward_func/std": 13.66856575012207, "step": 116 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 210.09375, "completions/mean_terminated_length": 140.42308044433594, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "epoch": 0.0936, "grad_norm": 12.358490943908691, "kl": 0.02017974853515625, "learning_rate": 1e-06, "loss": 0.0062, "num_tokens": 1487289.0, "reward": -5.93212890625, "reward_std": 5.732447147369385, "rewards/rm_reward_func/mean": -5.93212890625, "rewards/rm_reward_func/std": 10.107297897338867, "step": 117 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 479.0, "completions/mean_length": 307.0, "completions/mean_terminated_length": 184.0, "completions/min_length": 20.0, "completions/min_terminated_length": 20.0, "epoch": 0.0944, "grad_norm": 7.494678974151611, "kl": 0.012603759765625, "learning_rate": 1e-06, "loss": -0.1111, "num_tokens": 1501401.0, "reward": -0.51397705078125, "reward_std": 4.986804008483887, "rewards/rm_reward_func/mean": -0.51397705078125, "rewards/rm_reward_func/std": 6.126167297363281, "step": 118 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 221.59375, "completions/mean_terminated_length": 202.23333740234375, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "epoch": 0.0952, "grad_norm": 5.377990245819092, "kl": 0.0181732177734375, "learning_rate": 1e-06, "loss": -0.2771, "num_tokens": 1512988.0, "reward": -0.9085693359375, "reward_std": 4.9990386962890625, "rewards/rm_reward_func/mean": -0.9085693359375, "rewards/rm_reward_func/std": 6.47190523147583, "step": 119 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.4375, "completions/max_length": 512.0, "completions/max_terminated_length": 467.0, "completions/mean_length": 307.78125, "completions/mean_terminated_length": 148.94444274902344, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "epoch": 0.096, "grad_norm": 5.832651138305664, "kl": 0.01479339599609375, "learning_rate": 1e-06, "loss": 0.0782, "num_tokens": 1530557.0, "reward": -7.892578125, "reward_std": 3.8970372676849365, "rewards/rm_reward_func/mean": -7.892578125, "rewards/rm_reward_func/std": 6.989450454711914, "step": 120 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 454.0, "completions/mean_length": 275.4375, "completions/mean_terminated_length": 220.84616088867188, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "epoch": 0.0968, "grad_norm": 2.5381200313568115, "kl": 0.0143890380859375, "learning_rate": 1e-06, "loss": -0.1052, "num_tokens": 1542539.0, "reward": -1.7589111328125, "reward_std": 3.9134409427642822, "rewards/rm_reward_func/mean": -1.7589111328125, "rewards/rm_reward_func/std": 9.491743087768555, "step": 121 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40625, "completions/max_length": 512.0, "completions/max_terminated_length": 427.0, "completions/mean_length": 336.03125, "completions/mean_terminated_length": 215.63157653808594, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "epoch": 0.0976, "grad_norm": 4.144689559936523, "kl": 0.014801025390625, "learning_rate": 1e-06, "loss": 0.0091, "num_tokens": 1555908.0, "reward": -3.1043167114257812, "reward_std": 3.349581718444824, "rewards/rm_reward_func/mean": -3.1043167114257812, "rewards/rm_reward_func/std": 6.525447368621826, "step": 122 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 281.1875, "completions/mean_terminated_length": 257.3103332519531, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "epoch": 0.0984, "grad_norm": 10.447542190551758, "kl": 0.02520751953125, "learning_rate": 1e-06, "loss": 0.0106, "num_tokens": 1568394.0, "reward": 2.2360823154449463, "reward_std": 4.762975692749023, "rewards/rm_reward_func/mean": 2.2360823154449463, "rewards/rm_reward_func/std": 9.465394973754883, "step": 123 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 457.0, "completions/mean_length": 282.90625, "completions/mean_terminated_length": 240.48147583007812, "completions/min_length": 23.0, "completions/min_terminated_length": 23.0, "epoch": 0.0992, "grad_norm": 66.1051025390625, "kl": 0.0311279296875, "learning_rate": 1e-06, "loss": 0.0544, "num_tokens": 1584383.0, "reward": -4.30712890625, "reward_std": 7.795045852661133, "rewards/rm_reward_func/mean": -4.30712890625, "rewards/rm_reward_func/std": 11.063386917114258, "step": 124 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 470.0, "completions/mean_length": 276.84375, "completions/mean_terminated_length": 252.51724243164062, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "epoch": 0.1, "grad_norm": 3.615330457687378, "kl": 0.025054931640625, "learning_rate": 1e-06, "loss": 0.0065, "num_tokens": 1595322.0, "reward": -11.7393798828125, "reward_std": 2.689405918121338, "rewards/rm_reward_func/mean": -11.7393798828125, "rewards/rm_reward_func/std": 13.393643379211426, "step": 125 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 448.0, "completions/mean_length": 312.03125, "completions/mean_terminated_length": 207.2857208251953, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "epoch": 0.1008, "grad_norm": 23.150880813598633, "kl": 0.0230255126953125, "learning_rate": 1e-06, "loss": -0.1352, "num_tokens": 1609723.0, "reward": -8.863861083984375, "reward_std": 6.96150541305542, "rewards/rm_reward_func/mean": -8.863861083984375, "rewards/rm_reward_func/std": 9.106093406677246, "step": 126 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 392.78125, "completions/mean_terminated_length": 321.25, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "epoch": 0.1016, "grad_norm": 3.9794867038726807, "kl": 0.011749267578125, "learning_rate": 1e-06, "loss": 0.2709, "num_tokens": 1626636.0, "reward": -8.689453125, "reward_std": 4.927318572998047, "rewards/rm_reward_func/mean": -8.689453125, "rewards/rm_reward_func/std": 7.077872276306152, "step": 127 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 485.0, "completions/mean_length": 326.0, "completions/mean_terminated_length": 264.0, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "epoch": 0.1024, "grad_norm": 3.827911376953125, "kl": 0.0189971923828125, "learning_rate": 1e-06, "loss": -0.0156, "num_tokens": 1639084.0, "reward": -0.710693359375, "reward_std": 6.628046035766602, "rewards/rm_reward_func/mean": -0.710693359375, "rewards/rm_reward_func/std": 9.962486267089844, "step": 128 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 366.8125, "completions/mean_terminated_length": 310.0, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "epoch": 0.1032, "grad_norm": 2.512551784515381, "kl": 0.0133056640625, "learning_rate": 1e-06, "loss": 0.1508, "num_tokens": 1653750.0, "reward": 2.409698486328125, "reward_std": 8.444917678833008, "rewards/rm_reward_func/mean": 2.409698486328125, "rewards/rm_reward_func/std": 11.581578254699707, "step": 129 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 466.0, "completions/mean_length": 309.09375, "completions/mean_terminated_length": 271.5185241699219, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "epoch": 0.104, "grad_norm": 4.3505144119262695, "kl": 0.02154541015625, "learning_rate": 1e-06, "loss": -0.078, "num_tokens": 1666745.0, "reward": -0.958282470703125, "reward_std": 5.802244663238525, "rewards/rm_reward_func/mean": -0.958282470703125, "rewards/rm_reward_func/std": 7.622645378112793, "step": 130 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 390.0, "completions/mean_length": 206.59375, "completions/mean_terminated_length": 162.96429443359375, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "epoch": 0.1048, "grad_norm": 5.6502861976623535, "kl": 0.01812744140625, "learning_rate": 1e-06, "loss": 0.1992, "num_tokens": 1677804.0, "reward": -6.44091796875, "reward_std": 6.14309024810791, "rewards/rm_reward_func/mean": -6.44091796875, "rewards/rm_reward_func/std": 10.069220542907715, "step": 131 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 385.5, "completions/mean_terminated_length": 309.6000061035156, "completions/min_length": 92.0, "completions/min_terminated_length": 92.0, "epoch": 0.1056, "grad_norm": 2.8553056716918945, "kl": 0.02423095703125, "learning_rate": 1e-06, "loss": -0.1016, "num_tokens": 1693340.0, "reward": 2.6444091796875, "reward_std": 8.04133415222168, "rewards/rm_reward_func/mean": 2.6444091796875, "rewards/rm_reward_func/std": 10.943760871887207, "step": 132 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 495.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 324.65625, "completions/mean_terminated_length": 324.65625, "completions/min_length": 196.0, "completions/min_terminated_length": 196.0, "epoch": 0.1064, "grad_norm": 2.7388484477996826, "kl": 0.0200958251953125, "learning_rate": 1e-06, "loss": 0.0034, "num_tokens": 1705993.0, "reward": -2.5941162109375, "reward_std": 4.211709499359131, "rewards/rm_reward_func/mean": -2.5941162109375, "rewards/rm_reward_func/std": 6.4686055183410645, "step": 133 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 465.0, "completions/mean_length": 254.71875, "completions/mean_terminated_length": 207.07408142089844, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "epoch": 0.1072, "grad_norm": 4.826857089996338, "kl": 0.01715087890625, "learning_rate": 1e-06, "loss": 0.2512, "num_tokens": 1716728.0, "reward": -6.234710693359375, "reward_std": 6.557104110717773, "rewards/rm_reward_func/mean": -6.234710693359375, "rewards/rm_reward_func/std": 8.982573509216309, "step": 134 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 395.0, "completions/mean_length": 201.1875, "completions/mean_terminated_length": 191.16128540039062, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "epoch": 0.108, "grad_norm": 4.799185752868652, "kl": 0.0233612060546875, "learning_rate": 1e-06, "loss": -0.0819, "num_tokens": 1726414.0, "reward": -4.594329833984375, "reward_std": 5.226075649261475, "rewards/rm_reward_func/mean": -4.594329833984375, "rewards/rm_reward_func/std": 6.881073474884033, "step": 135 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 460.0, "completions/max_terminated_length": 460.0, "completions/mean_length": 223.96875, "completions/mean_terminated_length": 223.96875, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "epoch": 0.1088, "grad_norm": 9.70561408996582, "kl": 0.025299072265625, "learning_rate": 1e-06, "loss": 0.053, "num_tokens": 1736069.0, "reward": -0.31732177734375, "reward_std": 3.850168466567993, "rewards/rm_reward_func/mean": -0.31732177734375, "rewards/rm_reward_func/std": 7.10466194152832, "step": 136 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40625, "completions/max_length": 512.0, "completions/max_terminated_length": 477.0, "completions/mean_length": 329.125, "completions/mean_terminated_length": 204.0, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "epoch": 0.1096, "grad_norm": 2.982024669647217, "kl": 0.0196685791015625, "learning_rate": 1e-06, "loss": 0.0927, "num_tokens": 1749921.0, "reward": -6.221199035644531, "reward_std": 5.583423137664795, "rewards/rm_reward_func/mean": -6.221199035644531, "rewards/rm_reward_func/std": 7.993007659912109, "step": 137 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 457.0, "completions/mean_length": 215.84375, "completions/mean_terminated_length": 173.5357208251953, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "epoch": 0.1104, "grad_norm": 7.27449893951416, "kl": 0.0246734619140625, "learning_rate": 1e-06, "loss": 0.0318, "num_tokens": 1761748.0, "reward": -2.1171817779541016, "reward_std": 5.113093376159668, "rewards/rm_reward_func/mean": -2.1171817779541016, "rewards/rm_reward_func/std": 6.772409439086914, "step": 138 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 481.0, "completions/mean_length": 308.84375, "completions/mean_terminated_length": 241.125, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "epoch": 0.1112, "grad_norm": 5.053046226501465, "kl": 0.0249481201171875, "learning_rate": 1e-06, "loss": 0.0379, "num_tokens": 1775439.0, "reward": -1.70648193359375, "reward_std": 5.1250834465026855, "rewards/rm_reward_func/mean": -1.70648193359375, "rewards/rm_reward_func/std": 6.081603050231934, "step": 139 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 411.90625, "completions/mean_terminated_length": 351.8500061035156, "completions/min_length": 198.0, "completions/min_terminated_length": 198.0, "epoch": 0.112, "grad_norm": 2.3874924182891846, "kl": 0.0135955810546875, "learning_rate": 1e-06, "loss": 0.0458, "num_tokens": 1792004.0, "reward": 1.85693359375, "reward_std": 5.751528263092041, "rewards/rm_reward_func/mean": 1.85693359375, "rewards/rm_reward_func/std": 7.656973361968994, "step": 140 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 251.59375, "completions/mean_terminated_length": 234.2333526611328, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "epoch": 0.1128, "grad_norm": 3.846666097640991, "kl": 0.0225677490234375, "learning_rate": 1e-06, "loss": 0.103, "num_tokens": 1802663.0, "reward": -2.373046875, "reward_std": 6.561789035797119, "rewards/rm_reward_func/mean": -2.373046875, "rewards/rm_reward_func/std": 8.583110809326172, "step": 141 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 269.46875, "completions/mean_terminated_length": 213.50001525878906, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "epoch": 0.1136, "grad_norm": 3.5698037147521973, "kl": 0.03497314453125, "learning_rate": 1e-06, "loss": -0.0293, "num_tokens": 1814766.0, "reward": 0.341033935546875, "reward_std": 5.161829948425293, "rewards/rm_reward_func/mean": 0.341033935546875, "rewards/rm_reward_func/std": 7.019633769989014, "step": 142 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.46875, "completions/max_length": 512.0, "completions/max_terminated_length": 114.0, "completions/mean_length": 273.8125, "completions/mean_terminated_length": 63.64706039428711, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "epoch": 0.1144, "grad_norm": 9.236984252929688, "kl": 0.0199432373046875, "learning_rate": 1e-06, "loss": 0.3644, "num_tokens": 1827504.0, "reward": -11.159202575683594, "reward_std": 6.791610240936279, "rewards/rm_reward_func/mean": -11.159202575683594, "rewards/rm_reward_func/std": 8.779293060302734, "step": 143 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.75, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 476.4375, "completions/mean_terminated_length": 369.75, "completions/min_length": 124.0, "completions/min_terminated_length": 124.0, "epoch": 0.1152, "grad_norm": 2.5026986598968506, "kl": 0.0117034912109375, "learning_rate": 1e-06, "loss": -0.0064, "num_tokens": 1847678.0, "reward": -4.88604736328125, "reward_std": 4.872498035430908, "rewards/rm_reward_func/mean": -4.88604736328125, "rewards/rm_reward_func/std": 9.581089973449707, "step": 144 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 306.9375, "completions/mean_terminated_length": 268.96295166015625, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "epoch": 0.116, "grad_norm": 2.9737255573272705, "kl": 0.03043365478515625, "learning_rate": 1e-06, "loss": -0.0469, "num_tokens": 1863420.0, "reward": -2.0446929931640625, "reward_std": 3.8031177520751953, "rewards/rm_reward_func/mean": -2.0446929931640625, "rewards/rm_reward_func/std": 5.832355976104736, "step": 145 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.4375, "completions/max_length": 512.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 349.59375, "completions/mean_terminated_length": 223.2777862548828, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "epoch": 0.1168, "grad_norm": 4.583868503570557, "kl": 0.01976776123046875, "learning_rate": 1e-06, "loss": 0.0255, "num_tokens": 1876591.0, "reward": -1.48095703125, "reward_std": 4.935769081115723, "rewards/rm_reward_func/mean": -1.48095703125, "rewards/rm_reward_func/std": 16.34881019592285, "step": 146 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.4375, "completions/max_length": 512.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 407.25, "completions/mean_terminated_length": 325.77777099609375, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "epoch": 0.1176, "grad_norm": 3.11322021484375, "kl": 0.029876708984375, "learning_rate": 1e-06, "loss": -0.1804, "num_tokens": 1892183.0, "reward": -4.62982177734375, "reward_std": 6.981845855712891, "rewards/rm_reward_func/mean": -4.62982177734375, "rewards/rm_reward_func/std": 14.073089599609375, "step": 147 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 291.21875, "completions/mean_terminated_length": 240.2692413330078, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "epoch": 0.1184, "grad_norm": 3.6348979473114014, "kl": 0.02703857421875, "learning_rate": 1e-06, "loss": -0.0011, "num_tokens": 1905302.0, "reward": -3.182891845703125, "reward_std": 7.350401878356934, "rewards/rm_reward_func/mean": -3.182891845703125, "rewards/rm_reward_func/std": 10.496583938598633, "step": 148 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 206.125, "completions/mean_terminated_length": 174.48275756835938, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "epoch": 0.1192, "grad_norm": 6.732820987701416, "kl": 0.0248565673828125, "learning_rate": 1e-06, "loss": 0.0798, "num_tokens": 1915282.0, "reward": -6.91180419921875, "reward_std": 5.384483337402344, "rewards/rm_reward_func/mean": -6.91180419921875, "rewards/rm_reward_func/std": 8.211865425109863, "step": 149 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 371.78125, "completions/mean_terminated_length": 308.04547119140625, "completions/min_length": 161.0, "completions/min_terminated_length": 161.0, "epoch": 0.12, "grad_norm": 3.473301887512207, "kl": 0.02490234375, "learning_rate": 1e-06, "loss": -0.0287, "num_tokens": 1929515.0, "reward": -1.9688720703125, "reward_std": 5.15925931930542, "rewards/rm_reward_func/mean": -1.9688720703125, "rewards/rm_reward_func/std": 8.88673210144043, "step": 150 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 356.46875, "completions/mean_terminated_length": 334.25, "completions/min_length": 127.0, "completions/min_terminated_length": 127.0, "epoch": 0.1208, "grad_norm": 14.675175666809082, "kl": 0.0251617431640625, "learning_rate": 1e-06, "loss": -0.0117, "num_tokens": 1947602.0, "reward": 3.12274169921875, "reward_std": 5.56088399887085, "rewards/rm_reward_func/mean": 3.12274169921875, "rewards/rm_reward_func/std": 5.511745452880859, "step": 151 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 408.75, "completions/mean_terminated_length": 361.8182067871094, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "epoch": 0.1216, "grad_norm": 17.709932327270508, "kl": 0.0260009765625, "learning_rate": 1e-06, "loss": 0.1719, "num_tokens": 1964330.0, "reward": 1.937957763671875, "reward_std": 5.787369728088379, "rewards/rm_reward_func/mean": 1.937957763671875, "rewards/rm_reward_func/std": 7.468286037445068, "step": 152 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 412.78125, "completions/mean_terminated_length": 389.8846435546875, "completions/min_length": 112.0, "completions/min_terminated_length": 112.0, "epoch": 0.1224, "grad_norm": 3.4813315868377686, "kl": 0.027984619140625, "learning_rate": 1e-06, "loss": -0.0986, "num_tokens": 1981395.0, "reward": 0.11110877990722656, "reward_std": 4.848204612731934, "rewards/rm_reward_func/mean": 0.11110877990722656, "rewards/rm_reward_func/std": 8.962310791015625, "step": 153 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.4375, "completions/max_length": 512.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 384.375, "completions/mean_terminated_length": 285.1111145019531, "completions/min_length": 118.0, "completions/min_terminated_length": 118.0, "epoch": 0.1232, "grad_norm": 3.2246146202087402, "kl": 0.022918701171875, "learning_rate": 1e-06, "loss": 0.0176, "num_tokens": 1996591.0, "reward": 1.41229248046875, "reward_std": 7.246562957763672, "rewards/rm_reward_func/mean": 1.41229248046875, "rewards/rm_reward_func/std": 7.45635461807251, "step": 154 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 354.34375, "completions/mean_terminated_length": 325.1481628417969, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "epoch": 0.124, "grad_norm": 3.3056445121765137, "kl": 0.030120849609375, "learning_rate": 1e-06, "loss": -0.1102, "num_tokens": 2010946.0, "reward": 10.169677734375, "reward_std": 4.897927284240723, "rewards/rm_reward_func/mean": 10.169677734375, "rewards/rm_reward_func/std": 14.160859107971191, "step": 155 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 454.0, "completions/mean_length": 261.9375, "completions/mean_terminated_length": 215.629638671875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "epoch": 0.1248, "grad_norm": 27.496681213378906, "kl": 0.02239990234375, "learning_rate": 1e-06, "loss": -0.0082, "num_tokens": 2022800.0, "reward": -2.4217529296875, "reward_std": 6.2300496101379395, "rewards/rm_reward_func/mean": -2.4217529296875, "rewards/rm_reward_func/std": 10.626413345336914, "step": 156 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.46875, "completions/max_length": 512.0, "completions/max_terminated_length": 480.0, "completions/mean_length": 346.875, "completions/mean_terminated_length": 201.1764678955078, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "epoch": 0.1256, "grad_norm": 4.6230573654174805, "kl": 0.021440505981445312, "learning_rate": 1e-06, "loss": 0.1942, "num_tokens": 2036684.0, "reward": -7.7816162109375, "reward_std": 8.917112350463867, "rewards/rm_reward_func/mean": -7.7816162109375, "rewards/rm_reward_func/std": 14.053092956542969, "step": 157 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 392.03125, "completions/mean_terminated_length": 369.8148193359375, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "epoch": 0.1264, "grad_norm": 2.703683853149414, "kl": 0.034088134765625, "learning_rate": 1e-06, "loss": -0.0831, "num_tokens": 2051333.0, "reward": 0.690460205078125, "reward_std": 7.271797180175781, "rewards/rm_reward_func/mean": 0.690460205078125, "rewards/rm_reward_func/std": 11.116622924804688, "step": 158 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 486.0, "completions/mean_length": 276.875, "completions/mean_terminated_length": 184.86956787109375, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "epoch": 0.1272, "grad_norm": 7.96703577041626, "kl": 0.01966094970703125, "learning_rate": 1e-06, "loss": 0.2494, "num_tokens": 2063065.0, "reward": 5.6082763671875, "reward_std": 6.509998798370361, "rewards/rm_reward_func/mean": 5.6082763671875, "rewards/rm_reward_func/std": 8.432168960571289, "step": 159 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40625, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 380.1875, "completions/mean_terminated_length": 290.0, "completions/min_length": 121.0, "completions/min_terminated_length": 121.0, "epoch": 0.128, "grad_norm": 3.4324798583984375, "kl": 0.033443450927734375, "learning_rate": 1e-06, "loss": 0.0201, "num_tokens": 2077479.0, "reward": -6.3060302734375, "reward_std": 4.749573707580566, "rewards/rm_reward_func/mean": -6.3060302734375, "rewards/rm_reward_func/std": 13.343415260314941, "step": 160 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 350.25, "completions/mean_terminated_length": 296.3333435058594, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "epoch": 0.1288, "grad_norm": 2.8479349613189697, "kl": 0.023101806640625, "learning_rate": 1e-06, "loss": -0.0853, "num_tokens": 2091847.0, "reward": -2.8158111572265625, "reward_std": 3.8607840538024902, "rewards/rm_reward_func/mean": -2.8158111572265625, "rewards/rm_reward_func/std": 7.014427185058594, "step": 161 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 326.90625, "completions/mean_terminated_length": 242.77273559570312, "completions/min_length": 19.0, "completions/min_terminated_length": 19.0, "epoch": 0.1296, "grad_norm": 5.42257833480835, "kl": 0.03411865234375, "learning_rate": 1e-06, "loss": 0.025, "num_tokens": 2105620.0, "reward": -8.8624267578125, "reward_std": 3.4918594360351562, "rewards/rm_reward_func/mean": -8.8624267578125, "rewards/rm_reward_func/std": 6.637436866760254, "step": 162 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 482.0, "completions/mean_length": 197.5, "completions/mean_terminated_length": 164.96551513671875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "epoch": 0.1304, "grad_norm": 4.355756759643555, "kl": 0.037811279296875, "learning_rate": 1e-06, "loss": -0.0119, "num_tokens": 2114116.0, "reward": -11.6080322265625, "reward_std": 4.272393226623535, "rewards/rm_reward_func/mean": -11.6080322265625, "rewards/rm_reward_func/std": 6.417409896850586, "step": 163 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 276.53125, "completions/mean_terminated_length": 210.59999084472656, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "epoch": 0.1312, "grad_norm": 3.7714273929595947, "kl": 0.04034423828125, "learning_rate": 1e-06, "loss": 0.1036, "num_tokens": 2125205.0, "reward": -6.275434494018555, "reward_std": 6.214200973510742, "rewards/rm_reward_func/mean": -6.275434494018555, "rewards/rm_reward_func/std": 8.869131088256836, "step": 164 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 432.0, "completions/mean_length": 258.21875, "completions/mean_terminated_length": 241.30001831054688, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "epoch": 0.132, "grad_norm": 7.6665167808532715, "kl": 0.032684326171875, "learning_rate": 1e-06, "loss": -0.0494, "num_tokens": 2135652.0, "reward": -7.063232421875, "reward_std": 4.540660858154297, "rewards/rm_reward_func/mean": -7.063232421875, "rewards/rm_reward_func/std": 7.4130377769470215, "step": 165 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 366.40625, "completions/mean_terminated_length": 317.875, "completions/min_length": 139.0, "completions/min_terminated_length": 139.0, "epoch": 0.1328, "grad_norm": 2.8710451126098633, "kl": 0.0362548828125, "learning_rate": 1e-06, "loss": -0.0732, "num_tokens": 2149737.0, "reward": -1.594390869140625, "reward_std": 5.7335591316223145, "rewards/rm_reward_func/mean": -1.594390869140625, "rewards/rm_reward_func/std": 10.367554664611816, "step": 166 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 370.6875, "completions/mean_terminated_length": 315.39129638671875, "completions/min_length": 102.0, "completions/min_terminated_length": 102.0, "epoch": 0.1336, "grad_norm": 2.312666416168213, "kl": 0.01702880859375, "learning_rate": 1e-06, "loss": -0.113, "num_tokens": 2165135.0, "reward": 1.4799518585205078, "reward_std": 6.825066566467285, "rewards/rm_reward_func/mean": 1.4799518585205078, "rewards/rm_reward_func/std": 11.376241683959961, "step": 167 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 485.0, "completions/max_terminated_length": 485.0, "completions/mean_length": 189.3125, "completions/mean_terminated_length": 189.3125, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "epoch": 0.1344, "grad_norm": 11.69007396697998, "kl": 0.06195068359375, "learning_rate": 1e-06, "loss": 0.0645, "num_tokens": 2174529.0, "reward": 4.023796081542969, "reward_std": 5.484214782714844, "rewards/rm_reward_func/mean": 4.023796081542969, "rewards/rm_reward_func/std": 12.138504028320312, "step": 168 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 427.0, "completions/mean_length": 280.0, "completions/mean_terminated_length": 246.85714721679688, "completions/min_length": 105.0, "completions/min_terminated_length": 105.0, "epoch": 0.1352, "grad_norm": 3.6376917362213135, "kl": 0.037750244140625, "learning_rate": 1e-06, "loss": -0.0308, "num_tokens": 2185737.0, "reward": 3.1946182250976562, "reward_std": 4.653729438781738, "rewards/rm_reward_func/mean": 3.1946182250976562, "rewards/rm_reward_func/std": 8.615864753723145, "step": 169 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 475.0, "completions/mean_length": 288.53125, "completions/mean_terminated_length": 265.4137878417969, "completions/min_length": 83.0, "completions/min_terminated_length": 83.0, "epoch": 0.136, "grad_norm": 6.496363639831543, "kl": 0.042877197265625, "learning_rate": 1e-06, "loss": -0.1146, "num_tokens": 2197530.0, "reward": -3.2793121337890625, "reward_std": 5.921474456787109, "rewards/rm_reward_func/mean": -3.2793121337890625, "rewards/rm_reward_func/std": 9.333773612976074, "step": 170 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 293.46875, "completions/mean_terminated_length": 194.13636779785156, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "epoch": 0.1368, "grad_norm": 4.2207770347595215, "kl": 0.029510498046875, "learning_rate": 1e-06, "loss": 0.271, "num_tokens": 2210929.0, "reward": -6.6259765625, "reward_std": 6.164700984954834, "rewards/rm_reward_func/mean": -6.6259765625, "rewards/rm_reward_func/std": 8.203378677368164, "step": 171 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.46875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 425.0, "completions/mean_terminated_length": 348.23529052734375, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "epoch": 0.1376, "grad_norm": 2.683828592300415, "kl": 0.02593994140625, "learning_rate": 1e-06, "loss": 0.092, "num_tokens": 2229097.0, "reward": -2.466796875, "reward_std": 5.979694843292236, "rewards/rm_reward_func/mean": -2.466796875, "rewards/rm_reward_func/std": 21.459537506103516, "step": 172 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 312.15625, "completions/mean_terminated_length": 207.4761962890625, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "epoch": 0.1384, "grad_norm": 4.745087146759033, "kl": 0.027069091796875, "learning_rate": 1e-06, "loss": 0.2947, "num_tokens": 2244958.0, "reward": 0.202301025390625, "reward_std": 7.154237270355225, "rewards/rm_reward_func/mean": 0.202301025390625, "rewards/rm_reward_func/std": 12.688895225524902, "step": 173 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.53125, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 447.09375, "completions/mean_terminated_length": 373.5333557128906, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "epoch": 0.1392, "grad_norm": 11.718962669372559, "kl": 0.0233154296875, "learning_rate": 1e-06, "loss": -0.0934, "num_tokens": 2262017.0, "reward": 5.55755615234375, "reward_std": 6.223611354827881, "rewards/rm_reward_func/mean": 5.55755615234375, "rewards/rm_reward_func/std": 9.582088470458984, "step": 174 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 287.4375, "completions/mean_terminated_length": 280.19354248046875, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "epoch": 0.14, "grad_norm": 4.201067924499512, "kl": 0.060211181640625, "learning_rate": 1e-06, "loss": -0.0737, "num_tokens": 2275175.0, "reward": -4.629188537597656, "reward_std": 5.972632884979248, "rewards/rm_reward_func/mean": -4.629188537597656, "rewards/rm_reward_func/std": 9.620247840881348, "step": 175 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 315.46875, "completions/mean_terminated_length": 270.1153869628906, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "epoch": 0.1408, "grad_norm": 36.3863639831543, "kl": 0.034271240234375, "learning_rate": 1e-06, "loss": -0.0361, "num_tokens": 2288718.0, "reward": 2.839324951171875, "reward_std": 5.17338752746582, "rewards/rm_reward_func/mean": 2.839324951171875, "rewards/rm_reward_func/std": 6.9854583740234375, "step": 176 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.46875, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 459.3125, "completions/mean_terminated_length": 412.8235168457031, "completions/min_length": 284.0, "completions/min_terminated_length": 284.0, "epoch": 0.1416, "grad_norm": 2.37556529045105, "kl": 0.02569580078125, "learning_rate": 1e-06, "loss": -0.0249, "num_tokens": 2305264.0, "reward": 4.6612548828125, "reward_std": 4.835301399230957, "rewards/rm_reward_func/mean": 4.6612548828125, "rewards/rm_reward_func/std": 6.540696620941162, "step": 177 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 486.0, "completions/mean_length": 335.90625, "completions/mean_terminated_length": 277.2083435058594, "completions/min_length": 155.0, "completions/min_terminated_length": 155.0, "epoch": 0.1424, "grad_norm": 3.222507953643799, "kl": 0.023162841796875, "learning_rate": 1e-06, "loss": 0.0366, "num_tokens": 2320461.0, "reward": 0.611053466796875, "reward_std": 8.012971878051758, "rewards/rm_reward_func/mean": 0.611053466796875, "rewards/rm_reward_func/std": 21.763654708862305, "step": 178 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 301.0, "completions/mean_length": 276.0625, "completions/mean_terminated_length": 197.4166717529297, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "epoch": 0.1432, "grad_norm": 3.5161032676696777, "kl": 0.0360260009765625, "learning_rate": 1e-06, "loss": -0.0888, "num_tokens": 2331991.0, "reward": -10.7362060546875, "reward_std": 3.1551408767700195, "rewards/rm_reward_func/mean": -10.7362060546875, "rewards/rm_reward_func/std": 7.49362850189209, "step": 179 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 237.0625, "completions/mean_terminated_length": 145.4166717529297, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "epoch": 0.144, "grad_norm": 34.31532287597656, "kl": 0.02911376953125, "learning_rate": 1e-06, "loss": 0.3252, "num_tokens": 2342809.0, "reward": -6.35968017578125, "reward_std": 4.969329833984375, "rewards/rm_reward_func/mean": -6.35968017578125, "rewards/rm_reward_func/std": 10.100662231445312, "step": 180 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 444.0, "completions/max_terminated_length": 444.0, "completions/mean_length": 174.9375, "completions/mean_terminated_length": 174.9375, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "epoch": 0.1448, "grad_norm": 11.092037200927734, "kl": 0.059326171875, "learning_rate": 1e-06, "loss": -0.0305, "num_tokens": 2351343.0, "reward": 3.357177734375, "reward_std": 7.95058012008667, "rewards/rm_reward_func/mean": 3.357177734375, "rewards/rm_reward_func/std": 9.181958198547363, "step": 181 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 473.0, "completions/mean_length": 221.625, "completions/mean_terminated_length": 167.8518524169922, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "epoch": 0.1456, "grad_norm": 10.811847686767578, "kl": 0.04290771484375, "learning_rate": 1e-06, "loss": -0.1376, "num_tokens": 2362931.0, "reward": -2.74267578125, "reward_std": 5.7764482498168945, "rewards/rm_reward_func/mean": -2.74267578125, "rewards/rm_reward_func/std": 14.296358108520508, "step": 182 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 350.90625, "completions/mean_terminated_length": 266.5238037109375, "completions/min_length": 110.0, "completions/min_terminated_length": 110.0, "epoch": 0.1464, "grad_norm": 11.304616928100586, "kl": 0.02972412109375, "learning_rate": 1e-06, "loss": -0.0109, "num_tokens": 2376720.0, "reward": -2.41259765625, "reward_std": 4.233550071716309, "rewards/rm_reward_func/mean": -2.41259765625, "rewards/rm_reward_func/std": 8.830628395080566, "step": 183 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 388.46875, "completions/mean_terminated_length": 359.9615478515625, "completions/min_length": 237.0, "completions/min_terminated_length": 237.0, "epoch": 0.1472, "grad_norm": 17.827436447143555, "kl": 0.04998779296875, "learning_rate": 1e-06, "loss": -0.0211, "num_tokens": 2391807.0, "reward": 5.77008056640625, "reward_std": 6.17415189743042, "rewards/rm_reward_func/mean": 5.77008056640625, "rewards/rm_reward_func/std": 7.23347806930542, "step": 184 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 329.5, "completions/mean_terminated_length": 258.08697509765625, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "epoch": 0.148, "grad_norm": 20.43731117248535, "kl": 0.037933349609375, "learning_rate": 1e-06, "loss": 0.2097, "num_tokens": 2406519.0, "reward": -1.623046875, "reward_std": 10.683743476867676, "rewards/rm_reward_func/mean": -1.623046875, "rewards/rm_reward_func/std": 14.652113914489746, "step": 185 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 397.8125, "completions/mean_terminated_length": 345.9090881347656, "completions/min_length": 101.0, "completions/min_terminated_length": 101.0, "epoch": 0.1488, "grad_norm": 3.0334293842315674, "kl": 0.0233306884765625, "learning_rate": 1e-06, "loss": 0.0335, "num_tokens": 2421993.0, "reward": 8.384445190429688, "reward_std": 11.130109786987305, "rewards/rm_reward_func/mean": 8.384445190429688, "rewards/rm_reward_func/std": 12.353397369384766, "step": 186 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.46875, "completions/max_length": 512.0, "completions/max_terminated_length": 493.0, "completions/mean_length": 363.875, "completions/mean_terminated_length": 233.1764678955078, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "epoch": 0.1496, "grad_norm": 2.8142430782318115, "kl": 0.03497314453125, "learning_rate": 1e-06, "loss": 0.1352, "num_tokens": 2435917.0, "reward": -1.4100570678710938, "reward_std": 7.787389755249023, "rewards/rm_reward_func/mean": -1.4100570678710938, "rewards/rm_reward_func/std": 9.113836288452148, "step": 187 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 447.0, "completions/mean_length": 266.53125, "completions/mean_terminated_length": 250.16668701171875, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "epoch": 0.1504, "grad_norm": 8.15317440032959, "kl": 0.0330810546875, "learning_rate": 1e-06, "loss": -0.0168, "num_tokens": 2446750.0, "reward": -1.41357421875, "reward_std": 6.009856224060059, "rewards/rm_reward_func/mean": -1.41357421875, "rewards/rm_reward_func/std": 7.865617752075195, "step": 188 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 282.34375, "completions/mean_terminated_length": 162.04762268066406, "completions/min_length": 16.0, "completions/min_terminated_length": 16.0, "epoch": 0.1512, "grad_norm": 86.80154418945312, "kl": 0.05029296875, "learning_rate": 1e-06, "loss": 0.0314, "num_tokens": 2458649.0, "reward": -3.796173095703125, "reward_std": 4.965305328369141, "rewards/rm_reward_func/mean": -3.796173095703125, "rewards/rm_reward_func/std": 7.862216949462891, "step": 189 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 337.875, "completions/mean_terminated_length": 246.6666717529297, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "epoch": 0.152, "grad_norm": 3.135280132293701, "kl": 0.037353515625, "learning_rate": 1e-06, "loss": -0.1424, "num_tokens": 2471605.0, "reward": 4.255180358886719, "reward_std": 5.963335990905762, "rewards/rm_reward_func/mean": 4.255180358886719, "rewards/rm_reward_func/std": 7.225700855255127, "step": 190 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 327.96875, "completions/mean_terminated_length": 217.5500030517578, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "epoch": 0.1528, "grad_norm": 3.981858730316162, "kl": 0.0426025390625, "learning_rate": 1e-06, "loss": 0.187, "num_tokens": 2484356.0, "reward": -6.214874267578125, "reward_std": 4.893754482269287, "rewards/rm_reward_func/mean": -6.214874267578125, "rewards/rm_reward_func/std": 11.553343772888184, "step": 191 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 363.5625, "completions/mean_terminated_length": 305.478271484375, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "epoch": 0.1536, "grad_norm": 5.650016784667969, "kl": 0.03961181640625, "learning_rate": 1e-06, "loss": 0.2227, "num_tokens": 2499622.0, "reward": -3.061126708984375, "reward_std": 7.803011417388916, "rewards/rm_reward_func/mean": -3.061126708984375, "rewards/rm_reward_func/std": 11.103745460510254, "step": 192 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 282.375, "completions/mean_terminated_length": 239.8518524169922, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "epoch": 0.1544, "grad_norm": 4.78141975402832, "kl": 0.041748046875, "learning_rate": 1e-06, "loss": 0.0997, "num_tokens": 2511370.0, "reward": -4.6614990234375, "reward_std": 3.844748020172119, "rewards/rm_reward_func/mean": -4.6614990234375, "rewards/rm_reward_func/std": 6.409038066864014, "step": 193 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 218.28125, "completions/mean_terminated_length": 120.375, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "epoch": 0.1552, "grad_norm": 15.153929710388184, "kl": 0.03350830078125, "learning_rate": 1e-06, "loss": 0.1686, "num_tokens": 2524115.0, "reward": 1.6598663330078125, "reward_std": 6.3394975662231445, "rewards/rm_reward_func/mean": 1.6598663330078125, "rewards/rm_reward_func/std": 11.99753475189209, "step": 194 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 268.0, "completions/mean_length": 158.875, "completions/mean_terminated_length": 108.42857360839844, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "epoch": 0.156, "grad_norm": 7.861037731170654, "kl": 0.0665283203125, "learning_rate": 1e-06, "loss": -0.0664, "num_tokens": 2531239.0, "reward": -6.14581298828125, "reward_std": 6.690445423126221, "rewards/rm_reward_func/mean": -6.14581298828125, "rewards/rm_reward_func/std": 10.1651029586792, "step": 195 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 329.625, "completions/mean_terminated_length": 317.4666748046875, "completions/min_length": 152.0, "completions/min_terminated_length": 152.0, "epoch": 0.1568, "grad_norm": 8.710630416870117, "kl": 0.04339599609375, "learning_rate": 1e-06, "loss": 0.0139, "num_tokens": 2543811.0, "reward": 7.0421142578125, "reward_std": 7.355923175811768, "rewards/rm_reward_func/mean": 7.0421142578125, "rewards/rm_reward_func/std": 15.240861892700195, "step": 196 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 467.0, "completions/max_terminated_length": 467.0, "completions/mean_length": 174.8125, "completions/mean_terminated_length": 174.8125, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "epoch": 0.1576, "grad_norm": 4.834583759307861, "kl": 0.03778076171875, "learning_rate": 1e-06, "loss": 0.0676, "num_tokens": 2553517.0, "reward": 0.12786865234375, "reward_std": 3.288187265396118, "rewards/rm_reward_func/mean": 0.12786865234375, "rewards/rm_reward_func/std": 5.167597770690918, "step": 197 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 491.0, "completions/mean_length": 301.75, "completions/mean_terminated_length": 280.0, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "epoch": 0.1584, "grad_norm": 3.8911170959472656, "kl": 0.037628173828125, "learning_rate": 1e-06, "loss": 0.0381, "num_tokens": 2565581.0, "reward": 8.29638671875, "reward_std": 5.519439697265625, "rewards/rm_reward_func/mean": 8.29638671875, "rewards/rm_reward_func/std": 12.758336067199707, "step": 198 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 317.46875, "completions/mean_terminated_length": 252.625, "completions/min_length": 23.0, "completions/min_terminated_length": 23.0, "epoch": 0.1592, "grad_norm": 4.481633186340332, "kl": 0.0205078125, "learning_rate": 1e-06, "loss": 0.0399, "num_tokens": 2578404.0, "reward": -3.97637939453125, "reward_std": 6.747166633605957, "rewards/rm_reward_func/mean": -3.97637939453125, "rewards/rm_reward_func/std": 13.583218574523926, "step": 199 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.4375, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 334.21875, "completions/mean_terminated_length": 195.94444274902344, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "epoch": 0.16, "grad_norm": 2.933378219604492, "kl": 0.0440826416015625, "learning_rate": 1e-06, "loss": 0.0151, "num_tokens": 2591595.0, "reward": -2.1455078125, "reward_std": 2.6753859519958496, "rewards/rm_reward_func/mean": -2.1455078125, "rewards/rm_reward_func/std": 7.882194519042969, "step": 200 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 391.9375, "completions/mean_terminated_length": 358.3199768066406, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "epoch": 0.1608, "grad_norm": 4.800445079803467, "kl": 0.02825927734375, "learning_rate": 1e-06, "loss": 0.0658, "num_tokens": 2606889.0, "reward": 2.6326370239257812, "reward_std": 7.309446334838867, "rewards/rm_reward_func/mean": 2.6326370239257812, "rewards/rm_reward_func/std": 9.031261444091797, "step": 201 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 456.0, "completions/mean_length": 322.4375, "completions/mean_terminated_length": 287.3333435058594, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "epoch": 0.1616, "grad_norm": 6.8586320877075195, "kl": 0.07305908203125, "learning_rate": 1e-06, "loss": 0.0351, "num_tokens": 2623223.0, "reward": 0.64892578125, "reward_std": 3.7500052452087402, "rewards/rm_reward_func/mean": 0.64892578125, "rewards/rm_reward_func/std": 11.14384937286377, "step": 202 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 310.0, "completions/mean_terminated_length": 230.95652770996094, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "epoch": 0.1624, "grad_norm": 4.2218337059021, "kl": 0.039154052734375, "learning_rate": 1e-06, "loss": 0.1215, "num_tokens": 2638695.0, "reward": -3.1102294921875, "reward_std": 10.1006498336792, "rewards/rm_reward_func/mean": -3.1102294921875, "rewards/rm_reward_func/std": 10.23782730102539, "step": 203 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 470.0, "completions/mean_length": 271.6875, "completions/mean_terminated_length": 191.58334350585938, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "epoch": 0.1632, "grad_norm": 4.688625335693359, "kl": 0.0411376953125, "learning_rate": 1e-06, "loss": 0.2514, "num_tokens": 2652157.0, "reward": 2.2646484375, "reward_std": 7.363420486450195, "rewards/rm_reward_func/mean": 2.2646484375, "rewards/rm_reward_func/std": 11.030182838439941, "step": 204 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40625, "completions/max_length": 512.0, "completions/max_terminated_length": 473.0, "completions/mean_length": 306.90625, "completions/mean_terminated_length": 166.57894897460938, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "epoch": 0.164, "grad_norm": 4.533467769622803, "kl": 0.02496337890625, "learning_rate": 1e-06, "loss": -0.0195, "num_tokens": 2664242.0, "reward": -4.059173583984375, "reward_std": 5.695614337921143, "rewards/rm_reward_func/mean": -4.059173583984375, "rewards/rm_reward_func/std": 7.963764190673828, "step": 205 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 321.875, "completions/mean_terminated_length": 247.478271484375, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "epoch": 0.1648, "grad_norm": 4.341475009918213, "kl": 0.0380401611328125, "learning_rate": 1e-06, "loss": 0.0705, "num_tokens": 2677966.0, "reward": -6.3792724609375, "reward_std": 8.557640075683594, "rewards/rm_reward_func/mean": -6.3792724609375, "rewards/rm_reward_func/std": 10.657180786132812, "step": 206 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 254.1875, "completions/mean_terminated_length": 227.51724243164062, "completions/min_length": 23.0, "completions/min_terminated_length": 23.0, "epoch": 0.1656, "grad_norm": 18.00819206237793, "kl": 0.02423095703125, "learning_rate": 1e-06, "loss": -0.1429, "num_tokens": 2688732.0, "reward": -7.324462890625, "reward_std": 5.273411750793457, "rewards/rm_reward_func/mean": -7.324462890625, "rewards/rm_reward_func/std": 13.72179126739502, "step": 207 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 491.0, "completions/mean_length": 219.53125, "completions/mean_terminated_length": 200.03334045410156, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "epoch": 0.1664, "grad_norm": 9.372909545898438, "kl": 0.04443359375, "learning_rate": 1e-06, "loss": 0.1745, "num_tokens": 2700973.0, "reward": -3.869039535522461, "reward_std": 4.182497978210449, "rewards/rm_reward_func/mean": -3.869039535522461, "rewards/rm_reward_func/std": 5.315476417541504, "step": 208 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 381.875, "completions/mean_terminated_length": 313.71429443359375, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "epoch": 0.1672, "grad_norm": 14.11308765411377, "kl": 0.0384521484375, "learning_rate": 1e-06, "loss": -0.0134, "num_tokens": 2715521.0, "reward": 7.0210418701171875, "reward_std": 9.68466854095459, "rewards/rm_reward_func/mean": 7.0210418701171875, "rewards/rm_reward_func/std": 17.186697006225586, "step": 209 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 448.0, "completions/mean_length": 259.0, "completions/mean_terminated_length": 174.6666717529297, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "epoch": 0.168, "grad_norm": 4.891456604003906, "kl": 0.071197509765625, "learning_rate": 1e-06, "loss": -0.0402, "num_tokens": 2725753.0, "reward": -1.91473388671875, "reward_std": 5.077539920806885, "rewards/rm_reward_func/mean": -1.91473388671875, "rewards/rm_reward_func/std": 8.266571044921875, "step": 210 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.59375, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 413.21875, "completions/mean_terminated_length": 268.8461608886719, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "epoch": 0.1688, "grad_norm": 3.1043269634246826, "kl": 0.01953125, "learning_rate": 1e-06, "loss": 0.1178, "num_tokens": 2742696.0, "reward": -9.068359375, "reward_std": 7.59498405456543, "rewards/rm_reward_func/mean": -9.068359375, "rewards/rm_reward_func/std": 10.789223670959473, "step": 211 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40625, "completions/max_length": 512.0, "completions/max_terminated_length": 491.0, "completions/mean_length": 359.5, "completions/mean_terminated_length": 255.15789794921875, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "epoch": 0.1696, "grad_norm": 5.556434631347656, "kl": 0.03485107421875, "learning_rate": 1e-06, "loss": 0.0683, "num_tokens": 2758216.0, "reward": 2.6542510986328125, "reward_std": 4.50178337097168, "rewards/rm_reward_func/mean": 2.6542510986328125, "rewards/rm_reward_func/std": 7.546914100646973, "step": 212 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 250.25, "completions/mean_terminated_length": 223.1724090576172, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "epoch": 0.1704, "grad_norm": 4.8082275390625, "kl": 0.03082275390625, "learning_rate": 1e-06, "loss": 0.0538, "num_tokens": 2768344.0, "reward": 0.75689697265625, "reward_std": 5.467959880828857, "rewards/rm_reward_func/mean": 0.75689697265625, "rewards/rm_reward_func/std": 8.289709091186523, "step": 213 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 473.0, "completions/mean_length": 377.5625, "completions/mean_terminated_length": 307.1428527832031, "completions/min_length": 108.0, "completions/min_terminated_length": 108.0, "epoch": 0.1712, "grad_norm": 3.0155625343322754, "kl": 0.032958984375, "learning_rate": 1e-06, "loss": -0.0965, "num_tokens": 2782498.0, "reward": 0.0223388671875, "reward_std": 4.46353006362915, "rewards/rm_reward_func/mean": 0.0223388671875, "rewards/rm_reward_func/std": 11.193925857543945, "step": 214 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 203.6875, "completions/mean_terminated_length": 171.79310607910156, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "epoch": 0.172, "grad_norm": 5.511065483093262, "kl": 0.0542755126953125, "learning_rate": 1e-06, "loss": 0.3333, "num_tokens": 2792208.0, "reward": -2.86065673828125, "reward_std": 5.38389778137207, "rewards/rm_reward_func/mean": -2.86065673828125, "rewards/rm_reward_func/std": 11.51030158996582, "step": 215 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 397.84375, "completions/mean_terminated_length": 329.3500061035156, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "epoch": 0.1728, "grad_norm": 3.1028032302856445, "kl": 0.040283203125, "learning_rate": 1e-06, "loss": -0.0738, "num_tokens": 2807251.0, "reward": 5.0045166015625, "reward_std": 5.947755813598633, "rewards/rm_reward_func/mean": 5.0045166015625, "rewards/rm_reward_func/std": 6.689188480377197, "step": 216 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 473.0, "completions/mean_length": 267.875, "completions/mean_terminated_length": 199.51998901367188, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "epoch": 0.1736, "grad_norm": 12.485280990600586, "kl": 0.043365478515625, "learning_rate": 1e-06, "loss": -0.0573, "num_tokens": 2822775.0, "reward": -3.88580322265625, "reward_std": 5.136831283569336, "rewards/rm_reward_func/mean": -3.88580322265625, "rewards/rm_reward_func/std": 11.440207481384277, "step": 217 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 454.0, "completions/mean_length": 242.875, "completions/mean_terminated_length": 153.1666717529297, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "epoch": 0.1744, "grad_norm": 8.996060371398926, "kl": 0.0584716796875, "learning_rate": 1e-06, "loss": 0.3665, "num_tokens": 2833379.0, "reward": -2.662109375, "reward_std": 7.626079082489014, "rewards/rm_reward_func/mean": -2.662109375, "rewards/rm_reward_func/std": 9.784847259521484, "step": 218 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.53125, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 413.40625, "completions/mean_terminated_length": 301.66668701171875, "completions/min_length": 90.0, "completions/min_terminated_length": 90.0, "epoch": 0.1752, "grad_norm": 3.5558414459228516, "kl": 0.02426910400390625, "learning_rate": 1e-06, "loss": 0.0093, "num_tokens": 2849992.0, "reward": -1.936492919921875, "reward_std": 7.031126976013184, "rewards/rm_reward_func/mean": -1.936492919921875, "rewards/rm_reward_func/std": 12.036778450012207, "step": 219 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 357.40625, "completions/mean_terminated_length": 305.875, "completions/min_length": 91.0, "completions/min_terminated_length": 91.0, "epoch": 0.176, "grad_norm": 2.7269515991210938, "kl": 0.047210693359375, "learning_rate": 1e-06, "loss": 0.0937, "num_tokens": 2865101.0, "reward": 2.5566253662109375, "reward_std": 5.476223945617676, "rewards/rm_reward_func/mean": 2.5566253662109375, "rewards/rm_reward_func/std": 11.695185661315918, "step": 220 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 289.53125, "completions/mean_terminated_length": 274.70001220703125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "epoch": 0.1768, "grad_norm": 5.313303470611572, "kl": 0.05035400390625, "learning_rate": 1e-06, "loss": 0.295, "num_tokens": 2877494.0, "reward": 9.74462890625, "reward_std": 7.827642917633057, "rewards/rm_reward_func/mean": 9.74462890625, "rewards/rm_reward_func/std": 13.538431167602539, "step": 221 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 327.1875, "completions/mean_terminated_length": 254.86956787109375, "completions/min_length": 123.0, "completions/min_terminated_length": 123.0, "epoch": 0.1776, "grad_norm": 4.54102087020874, "kl": 0.0721435546875, "learning_rate": 1e-06, "loss": -0.0192, "num_tokens": 2890372.0, "reward": 4.88897705078125, "reward_std": 3.758185386657715, "rewards/rm_reward_func/mean": 4.88897705078125, "rewards/rm_reward_func/std": 7.395822525024414, "step": 222 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 263.65625, "completions/mean_terminated_length": 237.96551513671875, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "epoch": 0.1784, "grad_norm": 4.374941825866699, "kl": 0.082183837890625, "learning_rate": 1e-06, "loss": 0.0815, "num_tokens": 2901145.0, "reward": -4.588043212890625, "reward_std": 7.087510585784912, "rewards/rm_reward_func/mean": -4.588043212890625, "rewards/rm_reward_func/std": 14.971769332885742, "step": 223 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 441.0, "completions/mean_length": 228.15625, "completions/mean_terminated_length": 198.79310607910156, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "epoch": 0.1792, "grad_norm": 3.8707516193389893, "kl": 0.0631103515625, "learning_rate": 1e-06, "loss": -0.0906, "num_tokens": 2911422.0, "reward": 0.30474853515625, "reward_std": 6.519827842712402, "rewards/rm_reward_func/mean": 0.30474853515625, "rewards/rm_reward_func/std": 7.540925979614258, "step": 224 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 325.0, "completions/mean_length": 119.1875, "completions/mean_terminated_length": 106.51612854003906, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "epoch": 0.18, "grad_norm": 8.087342262268066, "kl": 0.083251953125, "learning_rate": 1e-06, "loss": 0.2222, "num_tokens": 2920028.0, "reward": -4.3948974609375, "reward_std": 5.361183166503906, "rewards/rm_reward_func/mean": -4.3948974609375, "rewards/rm_reward_func/std": 10.915719032287598, "step": 225 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 403.0, "completions/mean_length": 262.78125, "completions/mean_terminated_length": 205.2692413330078, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "epoch": 0.1808, "grad_norm": 5.448842525482178, "kl": 0.05535888671875, "learning_rate": 1e-06, "loss": 0.1111, "num_tokens": 2930717.0, "reward": 1.3837890625, "reward_std": 7.229766845703125, "rewards/rm_reward_func/mean": 1.3837890625, "rewards/rm_reward_func/std": 12.284672737121582, "step": 226 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 274.03125, "completions/mean_terminated_length": 240.0357208251953, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "epoch": 0.1816, "grad_norm": 5.66386604309082, "kl": 0.0574951171875, "learning_rate": 1e-06, "loss": 0.2228, "num_tokens": 2942294.0, "reward": -2.87408447265625, "reward_std": 8.044215202331543, "rewards/rm_reward_func/mean": -2.87408447265625, "rewards/rm_reward_func/std": 10.464455604553223, "step": 227 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 367.0, "completions/max_terminated_length": 367.0, "completions/mean_length": 184.78125, "completions/mean_terminated_length": 184.78125, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "epoch": 0.1824, "grad_norm": 6.5997467041015625, "kl": 0.0521240234375, "learning_rate": 1e-06, "loss": 0.0741, "num_tokens": 2953559.0, "reward": -5.8045654296875, "reward_std": 4.83258056640625, "rewards/rm_reward_func/mean": -5.8045654296875, "rewards/rm_reward_func/std": 9.505318641662598, "step": 228 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 293.0625, "completions/mean_terminated_length": 252.51852416992188, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "epoch": 0.1832, "grad_norm": 3.3603475093841553, "kl": 0.0386962890625, "learning_rate": 1e-06, "loss": 0.0101, "num_tokens": 2966665.0, "reward": -2.287109375, "reward_std": 4.542160987854004, "rewards/rm_reward_func/mean": -2.287109375, "rewards/rm_reward_func/std": 8.938611030578613, "step": 229 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 109.0, "completions/mean_length": 171.875, "completions/mean_terminated_length": 58.5, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "epoch": 0.184, "grad_norm": 5.757355213165283, "kl": 0.051544189453125, "learning_rate": 1e-06, "loss": -0.0941, "num_tokens": 2975517.0, "reward": -1.37554931640625, "reward_std": 6.1020684242248535, "rewards/rm_reward_func/mean": -1.37554931640625, "rewards/rm_reward_func/std": 9.515417098999023, "step": 230 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 373.875, "completions/mean_terminated_length": 319.8260803222656, "completions/min_length": 23.0, "completions/min_terminated_length": 23.0, "epoch": 0.1848, "grad_norm": 2.7484848499298096, "kl": 0.03887939453125, "learning_rate": 1e-06, "loss": -0.047, "num_tokens": 2989801.0, "reward": -6.4609375, "reward_std": 4.946435928344727, "rewards/rm_reward_func/mean": -6.4609375, "rewards/rm_reward_func/std": 7.9586381912231445, "step": 231 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 410.0, "completions/mean_length": 218.03125, "completions/mean_terminated_length": 163.59259033203125, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "epoch": 0.1856, "grad_norm": 4.303516864776611, "kl": 0.0555419921875, "learning_rate": 1e-06, "loss": 0.1213, "num_tokens": 2999034.0, "reward": -3.778125762939453, "reward_std": 6.124919891357422, "rewards/rm_reward_func/mean": -3.778125762939453, "rewards/rm_reward_func/std": 6.401005744934082, "step": 232 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 341.71875, "completions/mean_terminated_length": 324.10345458984375, "completions/min_length": 222.0, "completions/min_terminated_length": 222.0, "epoch": 0.1864, "grad_norm": 2.5016844272613525, "kl": 0.0266265869140625, "learning_rate": 1e-06, "loss": -0.0613, "num_tokens": 3012305.0, "reward": 0.1171875, "reward_std": 6.036954402923584, "rewards/rm_reward_func/mean": 0.1171875, "rewards/rm_reward_func/std": 8.2446870803833, "step": 233 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 325.0625, "completions/mean_terminated_length": 240.09091186523438, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "epoch": 0.1872, "grad_norm": 3.126168966293335, "kl": 0.033935546875, "learning_rate": 1e-06, "loss": 0.088, "num_tokens": 3024995.0, "reward": -2.0553665161132812, "reward_std": 4.089500427246094, "rewards/rm_reward_func/mean": -2.0553665161132812, "rewards/rm_reward_func/std": 6.073705196380615, "step": 234 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 348.03125, "completions/mean_terminated_length": 331.0689697265625, "completions/min_length": 181.0, "completions/min_terminated_length": 181.0, "epoch": 0.188, "grad_norm": 2.8671622276306152, "kl": 0.03570556640625, "learning_rate": 1e-06, "loss": -0.0666, "num_tokens": 3040716.0, "reward": 5.6767578125, "reward_std": 5.897713661193848, "rewards/rm_reward_func/mean": 5.6767578125, "rewards/rm_reward_func/std": 9.497367858886719, "step": 235 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 481.0, "completions/mean_length": 178.53125, "completions/mean_terminated_length": 144.03448486328125, "completions/min_length": 23.0, "completions/min_terminated_length": 23.0, "epoch": 0.1888, "grad_norm": 6.242638111114502, "kl": 0.05078125, "learning_rate": 1e-06, "loss": 0.0829, "num_tokens": 3050005.0, "reward": -10.707763671875, "reward_std": 5.651987552642822, "rewards/rm_reward_func/mean": -10.707763671875, "rewards/rm_reward_func/std": 7.248280048370361, "step": 236 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 501.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 233.0, "completions/mean_terminated_length": 233.0, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "epoch": 0.1896, "grad_norm": 3.2471871376037598, "kl": 0.02520751953125, "learning_rate": 1e-06, "loss": -0.0609, "num_tokens": 3060493.0, "reward": -0.5721435546875, "reward_std": 6.302328586578369, "rewards/rm_reward_func/mean": -0.5721435546875, "rewards/rm_reward_func/std": 10.102531433105469, "step": 237 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.625, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 453.0, "completions/mean_terminated_length": 354.66668701171875, "completions/min_length": 224.0, "completions/min_terminated_length": 224.0, "epoch": 0.1904, "grad_norm": 2.5366711616516113, "kl": 0.02734375, "learning_rate": 1e-06, "loss": 0.0203, "num_tokens": 3077549.0, "reward": -0.6923828125, "reward_std": 4.651421546936035, "rewards/rm_reward_func/mean": -0.6923828125, "rewards/rm_reward_func/std": 6.494197368621826, "step": 238 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 480.0, "completions/mean_length": 312.65625, "completions/mean_terminated_length": 266.65386962890625, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "epoch": 0.1912, "grad_norm": 3.5506675243377686, "kl": 0.038726806640625, "learning_rate": 1e-06, "loss": -0.0479, "num_tokens": 3090450.0, "reward": -0.1416015625, "reward_std": 5.028796672821045, "rewards/rm_reward_func/mean": -0.1416015625, "rewards/rm_reward_func/std": 7.697935581207275, "step": 239 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 342.75, "completions/mean_terminated_length": 265.81817626953125, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "epoch": 0.192, "grad_norm": 4.870519161224365, "kl": 0.02532958984375, "learning_rate": 1e-06, "loss": 0.0788, "num_tokens": 3105794.0, "reward": -0.5222930908203125, "reward_std": 7.223459243774414, "rewards/rm_reward_func/mean": -0.5222930908203125, "rewards/rm_reward_func/std": 8.785179138183594, "step": 240 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 459.0, "completions/mean_length": 328.71875, "completions/mean_terminated_length": 294.77777099609375, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "epoch": 0.1928, "grad_norm": 2.8499858379364014, "kl": 0.036407470703125, "learning_rate": 1e-06, "loss": -0.0227, "num_tokens": 3119169.0, "reward": -0.429351806640625, "reward_std": 4.58777379989624, "rewards/rm_reward_func/mean": -0.429351806640625, "rewards/rm_reward_func/std": 6.2184648513793945, "step": 241 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 512.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 417.875, "completions/mean_terminated_length": 323.75, "completions/min_length": 105.0, "completions/min_terminated_length": 105.0, "epoch": 0.1936, "grad_norm": 3.2108466625213623, "kl": 0.0294189453125, "learning_rate": 1e-06, "loss": 0.0375, "num_tokens": 3137933.0, "reward": -4.9796905517578125, "reward_std": 5.5578694343566895, "rewards/rm_reward_func/mean": -4.9796905517578125, "rewards/rm_reward_func/std": 14.907642364501953, "step": 242 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 485.0, "completions/mean_length": 221.375, "completions/mean_terminated_length": 167.55555725097656, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "epoch": 0.1944, "grad_norm": 4.164768695831299, "kl": 0.061614990234375, "learning_rate": 1e-06, "loss": -0.1704, "num_tokens": 3147577.0, "reward": -3.182586669921875, "reward_std": 5.680960655212402, "rewards/rm_reward_func/mean": -3.182586669921875, "rewards/rm_reward_func/std": 9.807361602783203, "step": 243 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 475.0, "completions/mean_length": 308.59375, "completions/mean_terminated_length": 270.9259338378906, "completions/min_length": 102.0, "completions/min_terminated_length": 102.0, "epoch": 0.1952, "grad_norm": 4.503950119018555, "kl": 0.0273284912109375, "learning_rate": 1e-06, "loss": -0.1201, "num_tokens": 3162324.0, "reward": -5.913942337036133, "reward_std": 5.996950626373291, "rewards/rm_reward_func/mean": -5.913942337036133, "rewards/rm_reward_func/std": 8.33896255493164, "step": 244 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 261.0, "completions/mean_terminated_length": 252.90321350097656, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "epoch": 0.196, "grad_norm": 32.10102081298828, "kl": 0.049560546875, "learning_rate": 1e-06, "loss": -0.0312, "num_tokens": 3176892.0, "reward": 5.21258544921875, "reward_std": 7.160707473754883, "rewards/rm_reward_func/mean": 5.21258544921875, "rewards/rm_reward_func/std": 9.029263496398926, "step": 245 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 350.0, "completions/mean_length": 174.59375, "completions/mean_terminated_length": 139.6896514892578, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "epoch": 0.1968, "grad_norm": 7.7096686363220215, "kl": 0.06005859375, "learning_rate": 1e-06, "loss": 0.4931, "num_tokens": 3185263.0, "reward": -0.90380859375, "reward_std": 8.060064315795898, "rewards/rm_reward_func/mean": -0.90380859375, "rewards/rm_reward_func/std": 9.48470687866211, "step": 246 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 474.0, "completions/mean_length": 252.09375, "completions/mean_terminated_length": 225.20689392089844, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "epoch": 0.1976, "grad_norm": 3.6530377864837646, "kl": 0.057586669921875, "learning_rate": 1e-06, "loss": -0.0903, "num_tokens": 3196322.0, "reward": -0.84429931640625, "reward_std": 3.671651840209961, "rewards/rm_reward_func/mean": -0.84429931640625, "rewards/rm_reward_func/std": 5.721739292144775, "step": 247 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 463.0, "completions/mean_length": 216.875, "completions/mean_terminated_length": 148.7692413330078, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "epoch": 0.1984, "grad_norm": 9.012897491455078, "kl": 0.060882568359375, "learning_rate": 1e-06, "loss": -0.0163, "num_tokens": 3206374.0, "reward": -1.3134765625, "reward_std": 1.9922480583190918, "rewards/rm_reward_func/mean": -1.3134765625, "rewards/rm_reward_func/std": 9.720147132873535, "step": 248 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 221.75, "completions/mean_terminated_length": 202.40000915527344, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "epoch": 0.1992, "grad_norm": 4.076515197753906, "kl": 0.05389404296875, "learning_rate": 1e-06, "loss": 0.0904, "num_tokens": 3217438.0, "reward": 5.044475555419922, "reward_std": 5.064403533935547, "rewards/rm_reward_func/mean": 5.044475555419922, "rewards/rm_reward_func/std": 11.514410972595215, "step": 249 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.4375, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 408.25, "completions/mean_terminated_length": 327.5555725097656, "completions/min_length": 110.0, "completions/min_terminated_length": 110.0, "epoch": 0.2, "grad_norm": 2.5681052207946777, "kl": 0.027313232421875, "learning_rate": 1e-06, "loss": 0.0815, "num_tokens": 3233350.0, "reward": -2.93603515625, "reward_std": 3.7280993461608887, "rewards/rm_reward_func/mean": -2.93603515625, "rewards/rm_reward_func/std": 12.866559028625488, "step": 250 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 336.5625, "completions/mean_terminated_length": 267.9130554199219, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "epoch": 0.2008, "grad_norm": 3.0375123023986816, "kl": 0.045684814453125, "learning_rate": 1e-06, "loss": -0.0519, "num_tokens": 3248000.0, "reward": -8.931640625, "reward_std": 7.065740585327148, "rewards/rm_reward_func/mean": -8.931640625, "rewards/rm_reward_func/std": 10.82422924041748, "step": 251 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 238.0, "completions/mean_terminated_length": 187.25926208496094, "completions/min_length": 10.0, "completions/min_terminated_length": 10.0, "epoch": 0.2016, "grad_norm": 5.324024677276611, "kl": 0.057220458984375, "learning_rate": 1e-06, "loss": 0.158, "num_tokens": 3259632.0, "reward": -5.93310546875, "reward_std": 4.27932071685791, "rewards/rm_reward_func/mean": -5.93310546875, "rewards/rm_reward_func/std": 12.716781616210938, "step": 252 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 321.0, "completions/mean_length": 147.09375, "completions/mean_terminated_length": 122.76667022705078, "completions/min_length": 16.0, "completions/min_terminated_length": 16.0, "epoch": 0.2024, "grad_norm": 4.747439861297607, "kl": 0.0233154296875, "learning_rate": 1e-06, "loss": -0.0654, "num_tokens": 3266675.0, "reward": -6.288330078125, "reward_std": 5.211904525756836, "rewards/rm_reward_func/mean": -6.288330078125, "rewards/rm_reward_func/std": 9.554975509643555, "step": 253 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 319.34375, "completions/mean_terminated_length": 283.6666564941406, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "epoch": 0.2032, "grad_norm": 3.346832513809204, "kl": 0.0314178466796875, "learning_rate": 1e-06, "loss": 0.0012, "num_tokens": 3280838.0, "reward": -1.9285697937011719, "reward_std": 6.478157043457031, "rewards/rm_reward_func/mean": -1.9285697937011719, "rewards/rm_reward_func/std": 10.842815399169922, "step": 254 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 365.46875, "completions/mean_terminated_length": 360.7419128417969, "completions/min_length": 205.0, "completions/min_terminated_length": 205.0, "epoch": 0.204, "grad_norm": 2.939826011657715, "kl": 0.04241943359375, "learning_rate": 1e-06, "loss": -0.0565, "num_tokens": 3294621.0, "reward": 10.9114990234375, "reward_std": 7.665554523468018, "rewards/rm_reward_func/mean": 10.9114990234375, "rewards/rm_reward_func/std": 11.80000114440918, "step": 255 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 510.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 238.4375, "completions/mean_terminated_length": 238.4375, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "epoch": 0.2048, "grad_norm": 3.935816764831543, "kl": 0.0316162109375, "learning_rate": 1e-06, "loss": -0.1044, "num_tokens": 3304547.0, "reward": -8.1998291015625, "reward_std": 5.198800086975098, "rewards/rm_reward_func/mean": -8.1998291015625, "rewards/rm_reward_func/std": 8.181034088134766, "step": 256 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 332.09375, "completions/mean_terminated_length": 313.4827575683594, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "epoch": 0.2056, "grad_norm": 3.129162311553955, "kl": 0.049652099609375, "learning_rate": 1e-06, "loss": 0.0076, "num_tokens": 3318494.0, "reward": 5.046142578125, "reward_std": 4.183533191680908, "rewards/rm_reward_func/mean": 5.046142578125, "rewards/rm_reward_func/std": 7.549215793609619, "step": 257 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 476.0, "completions/mean_length": 254.0, "completions/mean_terminated_length": 168.0, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "epoch": 0.2064, "grad_norm": 14.513141632080078, "kl": 0.0689849853515625, "learning_rate": 1e-06, "loss": -0.0949, "num_tokens": 3331886.0, "reward": -0.634033203125, "reward_std": 4.15193510055542, "rewards/rm_reward_func/mean": -0.634033203125, "rewards/rm_reward_func/std": 10.710453033447266, "step": 258 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 484.0, "completions/mean_length": 345.96875, "completions/mean_terminated_length": 290.625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "epoch": 0.2072, "grad_norm": 3.5183660984039307, "kl": 0.0648193359375, "learning_rate": 1e-06, "loss": 0.1997, "num_tokens": 3345461.0, "reward": 2.46270751953125, "reward_std": 8.18614673614502, "rewards/rm_reward_func/mean": 2.46270751953125, "rewards/rm_reward_func/std": 9.014900207519531, "step": 259 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 300.0, "completions/max_terminated_length": 300.0, "completions/mean_length": 181.75, "completions/mean_terminated_length": 181.75, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "epoch": 0.208, "grad_norm": 4.466091632843018, "kl": 0.07086181640625, "learning_rate": 1e-06, "loss": 0.0176, "num_tokens": 3356853.0, "reward": -4.78826904296875, "reward_std": 4.506010055541992, "rewards/rm_reward_func/mean": -4.78826904296875, "rewards/rm_reward_func/std": 9.68130111694336, "step": 260 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 436.0, "completions/mean_length": 277.53125, "completions/mean_terminated_length": 185.78260803222656, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "epoch": 0.2088, "grad_norm": 4.487113952636719, "kl": 0.04511070251464844, "learning_rate": 1e-06, "loss": -0.2554, "num_tokens": 3371294.0, "reward": -5.722412109375, "reward_std": 3.237701654434204, "rewards/rm_reward_func/mean": -5.722412109375, "rewards/rm_reward_func/std": 7.949067115783691, "step": 261 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 509.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 186.78125, "completions/mean_terminated_length": 186.78125, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "epoch": 0.2096, "grad_norm": 5.796142101287842, "kl": 0.08880615234375, "learning_rate": 1e-06, "loss": -0.1566, "num_tokens": 3379743.0, "reward": -4.72845458984375, "reward_std": 7.553511619567871, "rewards/rm_reward_func/mean": -4.72845458984375, "rewards/rm_reward_func/std": 11.902534484863281, "step": 262 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 199.625, "completions/mean_terminated_length": 141.7777862548828, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "epoch": 0.2104, "grad_norm": 7.342236042022705, "kl": 0.06292724609375, "learning_rate": 1e-06, "loss": -0.1884, "num_tokens": 3390747.0, "reward": 8.73486328125, "reward_std": 5.577756404876709, "rewards/rm_reward_func/mean": 8.73486328125, "rewards/rm_reward_func/std": 9.168144226074219, "step": 263 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 284.9375, "completions/mean_terminated_length": 242.88888549804688, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "epoch": 0.2112, "grad_norm": 4.4385223388671875, "kl": 0.0706787109375, "learning_rate": 1e-06, "loss": 0.1751, "num_tokens": 3401873.0, "reward": 4.10955810546875, "reward_std": 7.785256862640381, "rewards/rm_reward_func/mean": 4.10955810546875, "rewards/rm_reward_func/std": 9.873637199401855, "step": 264 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 309.5625, "completions/mean_terminated_length": 217.5454559326172, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "epoch": 0.212, "grad_norm": 4.364110469818115, "kl": 0.05230712890625, "learning_rate": 1e-06, "loss": -0.1079, "num_tokens": 3414907.0, "reward": -8.052734375, "reward_std": 4.879029273986816, "rewards/rm_reward_func/mean": -8.052734375, "rewards/rm_reward_func/std": 6.013888835906982, "step": 265 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 284.5, "completions/mean_terminated_length": 148.0, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "epoch": 0.2128, "grad_norm": 6.178150653839111, "kl": 0.0651397705078125, "learning_rate": 1e-06, "loss": 0.2324, "num_tokens": 3426979.0, "reward": -0.2777862548828125, "reward_std": 4.882081985473633, "rewards/rm_reward_func/mean": -0.2777862548828125, "rewards/rm_reward_func/std": 12.670929908752441, "step": 266 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 474.0, "completions/mean_length": 245.78125, "completions/mean_terminated_length": 171.239990234375, "completions/min_length": 10.0, "completions/min_terminated_length": 10.0, "epoch": 0.2136, "grad_norm": 7.719261169433594, "kl": 0.05291748046875, "learning_rate": 1e-06, "loss": 0.0783, "num_tokens": 3437388.0, "reward": -0.2825927734375, "reward_std": 5.749314308166504, "rewards/rm_reward_func/mean": -0.2825927734375, "rewards/rm_reward_func/std": 7.607658386230469, "step": 267 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 438.0, "completions/mean_length": 264.0, "completions/mean_terminated_length": 228.57144165039062, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "epoch": 0.2144, "grad_norm": 10.929339408874512, "kl": 0.05267333984375, "learning_rate": 1e-06, "loss": -0.0666, "num_tokens": 3450292.0, "reward": -1.218994140625, "reward_std": 3.6973087787628174, "rewards/rm_reward_func/mean": -1.218994140625, "rewards/rm_reward_func/std": 10.225122451782227, "step": 268 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 428.0, "completions/mean_length": 312.0, "completions/mean_terminated_length": 245.33334350585938, "completions/min_length": 104.0, "completions/min_terminated_length": 104.0, "epoch": 0.2152, "grad_norm": 3.8709096908569336, "kl": 0.03293609619140625, "learning_rate": 1e-06, "loss": 0.118, "num_tokens": 3463604.0, "reward": -4.028564453125, "reward_std": 5.055566787719727, "rewards/rm_reward_func/mean": -4.028564453125, "rewards/rm_reward_func/std": 13.777055740356445, "step": 269 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 382.0, "completions/mean_length": 154.9375, "completions/mean_terminated_length": 143.4193572998047, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "epoch": 0.216, "grad_norm": 4.488786220550537, "kl": 0.09271240234375, "learning_rate": 1e-06, "loss": -0.0088, "num_tokens": 3473114.0, "reward": 0.6698150634765625, "reward_std": 2.86566162109375, "rewards/rm_reward_func/mean": 0.6698150634765625, "rewards/rm_reward_func/std": 4.058006286621094, "step": 270 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 424.0, "completions/mean_length": 255.3125, "completions/mean_terminated_length": 169.75, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "epoch": 0.2168, "grad_norm": 6.03592586517334, "kl": 0.0821533203125, "learning_rate": 1e-06, "loss": -0.0379, "num_tokens": 3485300.0, "reward": -5.101806640625, "reward_std": 2.251300573348999, "rewards/rm_reward_func/mean": -5.101806640625, "rewards/rm_reward_func/std": 12.82911491394043, "step": 271 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 174.53125, "completions/mean_terminated_length": 163.64515686035156, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "epoch": 0.2176, "grad_norm": 5.9694294929504395, "kl": 0.07928466796875, "learning_rate": 1e-06, "loss": 0.1671, "num_tokens": 3495549.0, "reward": 1.263671875, "reward_std": 7.210736274719238, "rewards/rm_reward_func/mean": 1.263671875, "rewards/rm_reward_func/std": 12.690206527709961, "step": 272 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 275.59375, "completions/mean_terminated_length": 221.03846740722656, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "epoch": 0.2184, "grad_norm": 7.359996795654297, "kl": 0.0755615234375, "learning_rate": 1e-06, "loss": -0.0083, "num_tokens": 3510912.0, "reward": 9.77337646484375, "reward_std": 7.382805824279785, "rewards/rm_reward_func/mean": 9.77337646484375, "rewards/rm_reward_func/std": 12.175554275512695, "step": 273 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 482.0, "completions/mean_length": 209.53125, "completions/mean_terminated_length": 166.32144165039062, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "epoch": 0.2192, "grad_norm": 4.633665561676025, "kl": 0.06842041015625, "learning_rate": 1e-06, "loss": 0.0266, "num_tokens": 3521105.0, "reward": -6.2216796875, "reward_std": 11.70566177368164, "rewards/rm_reward_func/mean": -6.2216796875, "rewards/rm_reward_func/std": 14.380496978759766, "step": 274 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 308.375, "completions/mean_terminated_length": 251.36000061035156, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "epoch": 0.22, "grad_norm": 9.020089149475098, "kl": 0.05230712890625, "learning_rate": 1e-06, "loss": 0.0549, "num_tokens": 3534301.0, "reward": -1.356536865234375, "reward_std": 6.025744438171387, "rewards/rm_reward_func/mean": -1.356536865234375, "rewards/rm_reward_func/std": 11.09672737121582, "step": 275 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 215.46875, "completions/mean_terminated_length": 205.90321350097656, "completions/min_length": 102.0, "completions/min_terminated_length": 102.0, "epoch": 0.2208, "grad_norm": 3.9002256393432617, "kl": 0.039642333984375, "learning_rate": 1e-06, "loss": 0.0256, "num_tokens": 3544108.0, "reward": 1.199432373046875, "reward_std": 4.459294319152832, "rewards/rm_reward_func/mean": 1.199432373046875, "rewards/rm_reward_func/std": 5.9614973068237305, "step": 276 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 409.0, "completions/mean_terminated_length": 347.20001220703125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "epoch": 0.2216, "grad_norm": 2.783045768737793, "kl": 0.045440673828125, "learning_rate": 1e-06, "loss": -0.0599, "num_tokens": 3559324.0, "reward": 5.524658203125, "reward_std": 7.193890571594238, "rewards/rm_reward_func/mean": 5.524658203125, "rewards/rm_reward_func/std": 16.028470993041992, "step": 277 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 452.0, "completions/mean_length": 208.5625, "completions/mean_terminated_length": 177.1724090576172, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "epoch": 0.2224, "grad_norm": 4.956029415130615, "kl": 0.06280517578125, "learning_rate": 1e-06, "loss": -0.0151, "num_tokens": 3568326.0, "reward": -1.5927276611328125, "reward_std": 3.9969515800476074, "rewards/rm_reward_func/mean": -1.5927276611328125, "rewards/rm_reward_func/std": 9.469999313354492, "step": 278 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 401.0, "completions/mean_length": 261.375, "completions/mean_terminated_length": 147.4545440673828, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "epoch": 0.2232, "grad_norm": 3.7816061973571777, "kl": 0.062408447265625, "learning_rate": 1e-06, "loss": -0.2091, "num_tokens": 3579634.0, "reward": -7.9661865234375, "reward_std": 5.55967378616333, "rewards/rm_reward_func/mean": -7.9661865234375, "rewards/rm_reward_func/std": 10.989361763000488, "step": 279 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 475.0, "completions/mean_length": 314.46875, "completions/mean_terminated_length": 195.9499969482422, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "epoch": 0.224, "grad_norm": 3.4097347259521484, "kl": 0.06787109375, "learning_rate": 1e-06, "loss": -0.0373, "num_tokens": 3592225.0, "reward": -2.29248046875, "reward_std": 6.151001930236816, "rewards/rm_reward_func/mean": -2.29248046875, "rewards/rm_reward_func/std": 10.61819839477539, "step": 280 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 346.0, "completions/mean_length": 225.625, "completions/mean_terminated_length": 159.53846740722656, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "epoch": 0.2248, "grad_norm": 3.473419427871704, "kl": 0.08502197265625, "learning_rate": 1e-06, "loss": -0.0459, "num_tokens": 3601253.0, "reward": -2.1300048828125, "reward_std": 5.648019790649414, "rewards/rm_reward_func/mean": -2.1300048828125, "rewards/rm_reward_func/std": 6.02969217300415, "step": 281 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 227.625, "completions/mean_terminated_length": 148.0, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "epoch": 0.2256, "grad_norm": 14.872659683227539, "kl": 0.08856201171875, "learning_rate": 1e-06, "loss": -0.037, "num_tokens": 3615729.0, "reward": 3.3761444091796875, "reward_std": 2.2959201335906982, "rewards/rm_reward_func/mean": 3.3761444091796875, "rewards/rm_reward_func/std": 5.013526916503906, "step": 282 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 282.375, "completions/mean_terminated_length": 239.8518524169922, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "epoch": 0.2264, "grad_norm": 4.269347190856934, "kl": 0.05474853515625, "learning_rate": 1e-06, "loss": -0.0225, "num_tokens": 3626917.0, "reward": 5.6048431396484375, "reward_std": 4.628182411193848, "rewards/rm_reward_func/mean": 5.6048431396484375, "rewards/rm_reward_func/std": 8.509807586669922, "step": 283 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 266.0, "completions/max_terminated_length": 266.0, "completions/mean_length": 95.375, "completions/mean_terminated_length": 95.375, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "epoch": 0.2272, "grad_norm": 7.475558757781982, "kl": 0.13861083984375, "learning_rate": 1e-06, "loss": 0.0557, "num_tokens": 3633521.0, "reward": 1.9283256530761719, "reward_std": 1.3042426109313965, "rewards/rm_reward_func/mean": 1.9283256530761719, "rewards/rm_reward_func/std": 3.519742250442505, "step": 284 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 194.8125, "completions/mean_terminated_length": 121.61538696289062, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "epoch": 0.228, "grad_norm": 5.4495086669921875, "kl": 0.1224365234375, "learning_rate": 1e-06, "loss": -0.0286, "num_tokens": 3642939.0, "reward": 1.588470458984375, "reward_std": 2.9584450721740723, "rewards/rm_reward_func/mean": 1.588470458984375, "rewards/rm_reward_func/std": 8.191596031188965, "step": 285 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 224.5625, "completions/mean_terminated_length": 215.29031372070312, "completions/min_length": 110.0, "completions/min_terminated_length": 110.0, "epoch": 0.2288, "grad_norm": 3.4381351470947266, "kl": 0.0645751953125, "learning_rate": 1e-06, "loss": 0.0433, "num_tokens": 3653277.0, "reward": -6.94952392578125, "reward_std": 3.9081485271453857, "rewards/rm_reward_func/mean": -6.94952392578125, "rewards/rm_reward_func/std": 9.789095878601074, "step": 286 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 277.375, "completions/mean_terminated_length": 253.10345458984375, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "epoch": 0.2296, "grad_norm": 2.947699785232544, "kl": 0.07073974609375, "learning_rate": 1e-06, "loss": -0.0916, "num_tokens": 3664473.0, "reward": -3.729705810546875, "reward_std": 6.659585952758789, "rewards/rm_reward_func/mean": -3.729705810546875, "rewards/rm_reward_func/std": 9.191458702087402, "step": 287 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 453.0, "completions/mean_length": 217.15625, "completions/mean_terminated_length": 207.64515686035156, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "epoch": 0.2304, "grad_norm": 6.4276814460754395, "kl": 0.10601806640625, "learning_rate": 1e-06, "loss": 0.3546, "num_tokens": 3675502.0, "reward": 1.8604736328125, "reward_std": 5.1977338790893555, "rewards/rm_reward_func/mean": 1.8604736328125, "rewards/rm_reward_func/std": 6.9971723556518555, "step": 288 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 461.0, "completions/mean_length": 304.46875, "completions/mean_terminated_length": 235.2916717529297, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "epoch": 0.2312, "grad_norm": 3.1732778549194336, "kl": 0.0634765625, "learning_rate": 1e-06, "loss": -0.0507, "num_tokens": 3688813.0, "reward": 2.79443359375, "reward_std": 5.335002899169922, "rewards/rm_reward_func/mean": 2.79443359375, "rewards/rm_reward_func/std": 9.30492115020752, "step": 289 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 501.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 171.0625, "completions/mean_terminated_length": 171.0625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "epoch": 0.232, "grad_norm": 11.23501968383789, "kl": 0.07696533203125, "learning_rate": 1e-06, "loss": -0.0426, "num_tokens": 3698575.0, "reward": 7.9627685546875, "reward_std": 3.578082323074341, "rewards/rm_reward_func/mean": 7.9627685546875, "rewards/rm_reward_func/std": 10.838711738586426, "step": 290 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 315.625, "completions/mean_terminated_length": 238.78260803222656, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "epoch": 0.2328, "grad_norm": 5.33342981338501, "kl": 0.0753173828125, "learning_rate": 1e-06, "loss": -0.1215, "num_tokens": 3710707.0, "reward": 0.0675048828125, "reward_std": 6.095547199249268, "rewards/rm_reward_func/mean": 0.0675048828125, "rewards/rm_reward_func/std": 8.877041816711426, "step": 291 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 293.0, "completions/max_terminated_length": 293.0, "completions/mean_length": 101.1875, "completions/mean_terminated_length": 101.1875, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "epoch": 0.2336, "grad_norm": 4.05195951461792, "kl": 0.118408203125, "learning_rate": 1e-06, "loss": -0.0076, "num_tokens": 3717129.0, "reward": 2.620485305786133, "reward_std": 1.9676257371902466, "rewards/rm_reward_func/mean": 2.620485305786133, "rewards/rm_reward_func/std": 4.160001754760742, "step": 292 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 471.0, "completions/mean_length": 186.375, "completions/mean_terminated_length": 164.6666717529297, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "epoch": 0.2344, "grad_norm": 2.865370273590088, "kl": 0.09393310546875, "learning_rate": 1e-06, "loss": 0.1097, "num_tokens": 3725957.0, "reward": 1.41033935546875, "reward_std": 3.00130033493042, "rewards/rm_reward_func/mean": 1.41033935546875, "rewards/rm_reward_func/std": 9.897608757019043, "step": 293 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 328.1875, "completions/mean_terminated_length": 231.90476989746094, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "epoch": 0.2352, "grad_norm": 3.5064573287963867, "kl": 0.068359375, "learning_rate": 1e-06, "loss": -0.1071, "num_tokens": 3740923.0, "reward": -3.520263671875, "reward_std": 12.347634315490723, "rewards/rm_reward_func/mean": -3.520263671875, "rewards/rm_reward_func/std": 15.20130729675293, "step": 294 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 274.03125, "completions/mean_terminated_length": 194.70834350585938, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "epoch": 0.236, "grad_norm": 5.17435884475708, "kl": 0.0858154296875, "learning_rate": 1e-06, "loss": 0.2784, "num_tokens": 3751876.0, "reward": 0.765625, "reward_std": 6.9244384765625, "rewards/rm_reward_func/mean": 0.765625, "rewards/rm_reward_func/std": 14.027642250061035, "step": 295 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 200.25, "completions/mean_terminated_length": 155.71429443359375, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "epoch": 0.2368, "grad_norm": 6.697838306427002, "kl": 0.0771484375, "learning_rate": 1e-06, "loss": 0.232, "num_tokens": 3761100.0, "reward": 6.63671875, "reward_std": 5.848393440246582, "rewards/rm_reward_func/mean": 6.63671875, "rewards/rm_reward_func/std": 12.636683464050293, "step": 296 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 239.28125, "completions/mean_terminated_length": 162.9199981689453, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "epoch": 0.2376, "grad_norm": 2.993525743484497, "kl": 0.09796142578125, "learning_rate": 1e-06, "loss": -0.0469, "num_tokens": 3773725.0, "reward": 0.16357421875, "reward_std": 4.376680850982666, "rewards/rm_reward_func/mean": 0.16357421875, "rewards/rm_reward_func/std": 10.634230613708496, "step": 297 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 492.0, "completions/mean_length": 293.59375, "completions/mean_terminated_length": 279.0333557128906, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "epoch": 0.2384, "grad_norm": 7.341770172119141, "kl": 0.09283447265625, "learning_rate": 1e-06, "loss": 0.2676, "num_tokens": 3786040.0, "reward": 3.809326171875, "reward_std": 3.955899238586426, "rewards/rm_reward_func/mean": 3.809326171875, "rewards/rm_reward_func/std": 4.149126052856445, "step": 298 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 362.9375, "completions/mean_terminated_length": 328.5384826660156, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "epoch": 0.2392, "grad_norm": 2.8502986431121826, "kl": 0.05450439453125, "learning_rate": 1e-06, "loss": -0.0865, "num_tokens": 3800550.0, "reward": -2.6076202392578125, "reward_std": 6.401535987854004, "rewards/rm_reward_func/mean": -2.6076202392578125, "rewards/rm_reward_func/std": 12.927457809448242, "step": 299 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 333.0, "completions/mean_terminated_length": 299.85186767578125, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "epoch": 0.24, "grad_norm": 3.4198484420776367, "kl": 0.087646484375, "learning_rate": 1e-06, "loss": -0.0911, "num_tokens": 3813398.0, "reward": 10.27294921875, "reward_std": 8.61447525024414, "rewards/rm_reward_func/mean": 10.27294921875, "rewards/rm_reward_func/std": 12.064411163330078, "step": 300 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 315.0, "completions/mean_length": 226.9375, "completions/mean_terminated_length": 174.1481475830078, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "epoch": 0.2408, "grad_norm": 20.889190673828125, "kl": 0.064208984375, "learning_rate": 1e-06, "loss": 0.2754, "num_tokens": 3824964.0, "reward": 4.1443328857421875, "reward_std": 4.807868003845215, "rewards/rm_reward_func/mean": 4.1443328857421875, "rewards/rm_reward_func/std": 10.203561782836914, "step": 301 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 355.03125, "completions/mean_terminated_length": 293.60870361328125, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "epoch": 0.2416, "grad_norm": 3.480130910873413, "kl": 0.0916748046875, "learning_rate": 1e-06, "loss": 0.1964, "num_tokens": 3839741.0, "reward": 0.181640625, "reward_std": 6.749973773956299, "rewards/rm_reward_func/mean": 0.181640625, "rewards/rm_reward_func/std": 8.411787986755371, "step": 302 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 469.0, "completions/mean_length": 285.96875, "completions/mean_terminated_length": 167.57142639160156, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "epoch": 0.2424, "grad_norm": 5.95459508895874, "kl": 0.073211669921875, "learning_rate": 1e-06, "loss": 0.0174, "num_tokens": 3852788.0, "reward": -7.36474609375, "reward_std": 2.4573302268981934, "rewards/rm_reward_func/mean": -7.36474609375, "rewards/rm_reward_func/std": 10.699016571044922, "step": 303 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 474.0, "completions/mean_length": 293.40625, "completions/mean_terminated_length": 242.9615478515625, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "epoch": 0.2432, "grad_norm": 3.7543785572052, "kl": 0.0772705078125, "learning_rate": 1e-06, "loss": 0.2814, "num_tokens": 3864457.0, "reward": 4.44287109375, "reward_std": 8.432929039001465, "rewards/rm_reward_func/mean": 4.44287109375, "rewards/rm_reward_func/std": 20.573001861572266, "step": 304 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 198.5, "completions/mean_terminated_length": 188.3870849609375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "epoch": 0.244, "grad_norm": 3.925734281539917, "kl": 0.0897216796875, "learning_rate": 1e-06, "loss": 0.0502, "num_tokens": 3874617.0, "reward": -5.2659912109375, "reward_std": 3.777980327606201, "rewards/rm_reward_func/mean": -5.2659912109375, "rewards/rm_reward_func/std": 7.426268577575684, "step": 305 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 472.0, "completions/mean_length": 257.0, "completions/mean_terminated_length": 248.77418518066406, "completions/min_length": 128.0, "completions/min_terminated_length": 128.0, "epoch": 0.2448, "grad_norm": 5.314541816711426, "kl": 0.0880126953125, "learning_rate": 1e-06, "loss": -0.1034, "num_tokens": 3885481.0, "reward": -1.97320556640625, "reward_std": 4.8526153564453125, "rewards/rm_reward_func/mean": -1.97320556640625, "rewards/rm_reward_func/std": 8.424205780029297, "step": 306 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 464.0, "completions/max_terminated_length": 464.0, "completions/mean_length": 241.5, "completions/mean_terminated_length": 241.5, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "epoch": 0.2456, "grad_norm": 3.4764163494110107, "kl": 0.1007080078125, "learning_rate": 1e-06, "loss": 0.0276, "num_tokens": 3895369.0, "reward": 14.22119140625, "reward_std": 3.785148859024048, "rewards/rm_reward_func/mean": 14.22119140625, "rewards/rm_reward_func/std": 10.576017379760742, "step": 307 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 371.0, "completions/mean_length": 297.125, "completions/mean_terminated_length": 225.5, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "epoch": 0.2464, "grad_norm": 3.061163902282715, "kl": 0.042938232421875, "learning_rate": 1e-06, "loss": 0.0031, "num_tokens": 3907725.0, "reward": -2.1278076171875, "reward_std": 2.4199328422546387, "rewards/rm_reward_func/mean": -2.1278076171875, "rewards/rm_reward_func/std": 5.059024333953857, "step": 308 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 461.0, "completions/mean_length": 273.6875, "completions/mean_terminated_length": 266.0, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "epoch": 0.2472, "grad_norm": 3.6154208183288574, "kl": 0.08709716796875, "learning_rate": 1e-06, "loss": -0.0281, "num_tokens": 3918539.0, "reward": 5.895008087158203, "reward_std": 3.205272912979126, "rewards/rm_reward_func/mean": 5.895008087158203, "rewards/rm_reward_func/std": 7.15859317779541, "step": 309 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 410.0, "completions/mean_length": 283.6875, "completions/mean_terminated_length": 231.00001525878906, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "epoch": 0.248, "grad_norm": 2.8838815689086914, "kl": 0.035888671875, "learning_rate": 1e-06, "loss": -0.0877, "num_tokens": 3932985.0, "reward": -2.539306640625, "reward_std": 4.815282344818115, "rewards/rm_reward_func/mean": -2.539306640625, "rewards/rm_reward_func/std": 8.142019271850586, "step": 310 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 295.1875, "completions/mean_terminated_length": 288.19354248046875, "completions/min_length": 122.0, "completions/min_terminated_length": 122.0, "epoch": 0.2488, "grad_norm": 3.6053807735443115, "kl": 0.07330322265625, "learning_rate": 1e-06, "loss": -0.0069, "num_tokens": 3944983.0, "reward": 5.669269561767578, "reward_std": 5.6011061668396, "rewards/rm_reward_func/mean": 5.669269561767578, "rewards/rm_reward_func/std": 10.164783477783203, "step": 311 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 277.4375, "completions/mean_terminated_length": 261.8000183105469, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "epoch": 0.2496, "grad_norm": 4.000564098358154, "kl": 0.091552734375, "learning_rate": 1e-06, "loss": 0.0693, "num_tokens": 3957221.0, "reward": 6.09881591796875, "reward_std": 6.87520170211792, "rewards/rm_reward_func/mean": 6.09881591796875, "rewards/rm_reward_func/std": 14.669084548950195, "step": 312 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 439.0, "completions/mean_length": 151.09375, "completions/mean_terminated_length": 113.75862121582031, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "epoch": 0.2504, "grad_norm": 3.9963266849517822, "kl": 0.10784912109375, "learning_rate": 1e-06, "loss": -0.0074, "num_tokens": 3966792.0, "reward": 2.5313186645507812, "reward_std": 3.022167682647705, "rewards/rm_reward_func/mean": 2.5313186645507812, "rewards/rm_reward_func/std": 6.7323784828186035, "step": 313 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 353.5, "completions/mean_terminated_length": 300.66668701171875, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "epoch": 0.2512, "grad_norm": 7.695732116699219, "kl": 0.05596923828125, "learning_rate": 1e-06, "loss": -0.145, "num_tokens": 3980432.0, "reward": -0.392333984375, "reward_std": 5.235438346862793, "rewards/rm_reward_func/mean": -0.392333984375, "rewards/rm_reward_func/std": 11.274943351745605, "step": 314 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 447.0, "completions/max_terminated_length": 447.0, "completions/mean_length": 226.90625, "completions/mean_terminated_length": 226.90625, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "epoch": 0.252, "grad_norm": 19.040664672851562, "kl": 0.09423828125, "learning_rate": 1e-06, "loss": -0.0073, "num_tokens": 3993069.0, "reward": 3.00018310546875, "reward_std": 2.8838400840759277, "rewards/rm_reward_func/mean": 3.00018310546875, "rewards/rm_reward_func/std": 4.894301414489746, "step": 315 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 479.0, "completions/mean_length": 191.78125, "completions/mean_terminated_length": 181.4516143798828, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "epoch": 0.2528, "grad_norm": 12.309869766235352, "kl": 0.119140625, "learning_rate": 1e-06, "loss": 0.2599, "num_tokens": 4004622.0, "reward": 5.98388671875, "reward_std": 5.505092144012451, "rewards/rm_reward_func/mean": 5.98388671875, "rewards/rm_reward_func/std": 7.081192970275879, "step": 316 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 239.34375, "completions/mean_terminated_length": 211.13792419433594, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "epoch": 0.2536, "grad_norm": 5.658858299255371, "kl": 0.09930419921875, "learning_rate": 1e-06, "loss": -0.0048, "num_tokens": 4015681.0, "reward": -1.23388671875, "reward_std": 4.671802520751953, "rewards/rm_reward_func/mean": -1.23388671875, "rewards/rm_reward_func/std": 10.551041603088379, "step": 317 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 251.28125, "completions/mean_terminated_length": 191.11538696289062, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "epoch": 0.2544, "grad_norm": 3.543397903442383, "kl": 0.0836181640625, "learning_rate": 1e-06, "loss": -0.0158, "num_tokens": 4026234.0, "reward": -0.675323486328125, "reward_std": 2.8310508728027344, "rewards/rm_reward_func/mean": -0.675323486328125, "rewards/rm_reward_func/std": 6.245419502258301, "step": 318 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 178.8125, "completions/mean_terminated_length": 178.8125, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "epoch": 0.2552, "grad_norm": 4.103335380554199, "kl": 0.115234375, "learning_rate": 1e-06, "loss": 0.049, "num_tokens": 4035396.0, "reward": 3.646728515625, "reward_std": 3.5850768089294434, "rewards/rm_reward_func/mean": 3.646728515625, "rewards/rm_reward_func/std": 9.643186569213867, "step": 319 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 229.0, "completions/max_terminated_length": 229.0, "completions/mean_length": 116.25, "completions/mean_terminated_length": 116.25, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "epoch": 0.256, "grad_norm": 3.8824329376220703, "kl": 0.14990234375, "learning_rate": 1e-06, "loss": -0.0005, "num_tokens": 4041572.0, "reward": -1.460205078125, "reward_std": 2.3325414657592773, "rewards/rm_reward_func/mean": -1.460205078125, "rewards/rm_reward_func/std": 8.575539588928223, "step": 320 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 319.9375, "completions/mean_terminated_length": 255.9166717529297, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "epoch": 0.2568, "grad_norm": 8.91545295715332, "kl": 0.07806396484375, "learning_rate": 1e-06, "loss": 0.0173, "num_tokens": 4054298.0, "reward": 2.38446044921875, "reward_std": 3.706024169921875, "rewards/rm_reward_func/mean": 2.38446044921875, "rewards/rm_reward_func/std": 6.616323471069336, "step": 321 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 269.375, "completions/mean_terminated_length": 224.44444274902344, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "epoch": 0.2576, "grad_norm": 4.994580268859863, "kl": 0.15380859375, "learning_rate": 1e-06, "loss": 0.0178, "num_tokens": 4065390.0, "reward": 7.4681549072265625, "reward_std": 3.772608518600464, "rewards/rm_reward_func/mean": 7.4681549072265625, "rewards/rm_reward_func/std": 8.73366641998291, "step": 322 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 306.84375, "completions/mean_terminated_length": 238.45834350585938, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "epoch": 0.2584, "grad_norm": 6.513478755950928, "kl": 0.08917236328125, "learning_rate": 1e-06, "loss": 0.4623, "num_tokens": 4079945.0, "reward": 1.5455856323242188, "reward_std": 8.046390533447266, "rewards/rm_reward_func/mean": 1.5455856323242188, "rewards/rm_reward_func/std": 9.18606185913086, "step": 323 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 308.5625, "completions/mean_terminated_length": 295.0000305175781, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "epoch": 0.2592, "grad_norm": 3.7793478965759277, "kl": 0.08447265625, "learning_rate": 1e-06, "loss": -0.0834, "num_tokens": 4091579.0, "reward": 2.5294189453125, "reward_std": 5.412877559661865, "rewards/rm_reward_func/mean": 2.5294189453125, "rewards/rm_reward_func/std": 6.231657028198242, "step": 324 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 264.53125, "completions/mean_terminated_length": 218.70370483398438, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "epoch": 0.26, "grad_norm": 15.98507022857666, "kl": 0.10076904296875, "learning_rate": 1e-06, "loss": 0.2493, "num_tokens": 4105692.0, "reward": 3.5554351806640625, "reward_std": 5.774120330810547, "rewards/rm_reward_func/mean": 3.5554351806640625, "rewards/rm_reward_func/std": 7.647456169128418, "step": 325 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 323.625, "completions/mean_terminated_length": 304.137939453125, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "epoch": 0.2608, "grad_norm": 5.084076881408691, "kl": 0.070556640625, "learning_rate": 1e-06, "loss": -0.157, "num_tokens": 4119744.0, "reward": 6.6622314453125, "reward_std": 5.713457107543945, "rewards/rm_reward_func/mean": 6.6622314453125, "rewards/rm_reward_func/std": 9.382472038269043, "step": 326 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 469.0, "completions/mean_length": 380.6875, "completions/mean_terminated_length": 301.8999938964844, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "epoch": 0.2616, "grad_norm": 7.414204120635986, "kl": 0.09124755859375, "learning_rate": 1e-06, "loss": -0.1516, "num_tokens": 4134790.0, "reward": -4.882232666015625, "reward_std": 4.5240888595581055, "rewards/rm_reward_func/mean": -4.882232666015625, "rewards/rm_reward_func/std": 9.277235984802246, "step": 327 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 295.09375, "completions/mean_terminated_length": 254.92593383789062, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "epoch": 0.2624, "grad_norm": 25.824256896972656, "kl": 0.07318115234375, "learning_rate": 1e-06, "loss": 0.0039, "num_tokens": 4149569.0, "reward": 5.16973876953125, "reward_std": 6.8412652015686035, "rewards/rm_reward_func/mean": 5.16973876953125, "rewards/rm_reward_func/std": 9.498034477233887, "step": 328 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 492.0, "completions/mean_length": 410.5, "completions/mean_terminated_length": 364.3636474609375, "completions/min_length": 157.0, "completions/min_terminated_length": 157.0, "epoch": 0.2632, "grad_norm": 2.756131172180176, "kl": 0.0433349609375, "learning_rate": 1e-06, "loss": -0.1067, "num_tokens": 4164521.0, "reward": 0.5872802734375, "reward_std": 5.443880558013916, "rewards/rm_reward_func/mean": 0.5872802734375, "rewards/rm_reward_func/std": 6.557459354400635, "step": 329 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 302.65625, "completions/mean_terminated_length": 263.8888854980469, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "epoch": 0.264, "grad_norm": 5.988248348236084, "kl": 0.10546875, "learning_rate": 1e-06, "loss": -0.078, "num_tokens": 4176206.0, "reward": -2.9931640625, "reward_std": 5.075013160705566, "rewards/rm_reward_func/mean": -2.9931640625, "rewards/rm_reward_func/std": 7.403041839599609, "step": 330 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 316.4375, "completions/mean_terminated_length": 227.5454559326172, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "epoch": 0.2648, "grad_norm": 8.037940979003906, "kl": 0.087493896484375, "learning_rate": 1e-06, "loss": -0.0889, "num_tokens": 4188604.0, "reward": -0.03961181640625, "reward_std": 5.210146903991699, "rewards/rm_reward_func/mean": -0.03961181640625, "rewards/rm_reward_func/std": 6.811439514160156, "step": 331 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 352.6875, "completions/mean_terminated_length": 342.0666809082031, "completions/min_length": 137.0, "completions/min_terminated_length": 137.0, "epoch": 0.2656, "grad_norm": 3.725177049636841, "kl": 0.088134765625, "learning_rate": 1e-06, "loss": -0.0404, "num_tokens": 4201738.0, "reward": 2.9521484375, "reward_std": 6.857073783874512, "rewards/rm_reward_func/mean": 2.9521484375, "rewards/rm_reward_func/std": 7.544101238250732, "step": 332 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 386.8125, "completions/mean_terminated_length": 345.0833435058594, "completions/min_length": 197.0, "completions/min_terminated_length": 197.0, "epoch": 0.2664, "grad_norm": 2.696730613708496, "kl": 0.084686279296875, "learning_rate": 1e-06, "loss": -0.0032, "num_tokens": 4216772.0, "reward": 5.23828125, "reward_std": 5.675407409667969, "rewards/rm_reward_func/mean": 5.23828125, "rewards/rm_reward_func/std": 6.8515472412109375, "step": 333 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 417.0, "completions/mean_length": 214.78125, "completions/mean_terminated_length": 205.19354248046875, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "epoch": 0.2672, "grad_norm": 4.98583459854126, "kl": 0.0863037109375, "learning_rate": 1e-06, "loss": -0.0115, "num_tokens": 4229349.0, "reward": -2.3466796875, "reward_std": 5.685585021972656, "rewards/rm_reward_func/mean": -2.3466796875, "rewards/rm_reward_func/std": 7.332208633422852, "step": 334 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.625, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 408.78125, "completions/mean_terminated_length": 236.75, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "epoch": 0.268, "grad_norm": 3.633908987045288, "kl": 0.0657958984375, "learning_rate": 1e-06, "loss": 0.0563, "num_tokens": 4247326.0, "reward": 1.4497833251953125, "reward_std": 5.859368801116943, "rewards/rm_reward_func/mean": 1.4497833251953125, "rewards/rm_reward_func/std": 7.945054054260254, "step": 335 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 398.375, "completions/mean_terminated_length": 377.3333435058594, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "epoch": 0.2688, "grad_norm": 6.791285514831543, "kl": 0.0994873046875, "learning_rate": 1e-06, "loss": -0.0843, "num_tokens": 4263082.0, "reward": 1.28485107421875, "reward_std": 5.6787285804748535, "rewards/rm_reward_func/mean": 1.28485107421875, "rewards/rm_reward_func/std": 12.219388961791992, "step": 336 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 452.0, "completions/mean_length": 312.9375, "completions/mean_terminated_length": 299.66668701171875, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.2696, "grad_norm": 3.929457902908325, "kl": 0.0523681640625, "learning_rate": 1e-06, "loss": 0.014, "num_tokens": 4275208.0, "reward": 7.05194091796875, "reward_std": 5.769165992736816, "rewards/rm_reward_func/mean": 7.05194091796875, "rewards/rm_reward_func/std": 16.998645782470703, "step": 337 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 456.34375, "completions/mean_terminated_length": 422.95001220703125, "completions/min_length": 213.0, "completions/min_terminated_length": 213.0, "epoch": 0.2704, "grad_norm": 4.277588844299316, "kl": 0.04718017578125, "learning_rate": 1e-06, "loss": -0.0212, "num_tokens": 4293051.0, "reward": -1.598388671875, "reward_std": 5.49388313293457, "rewards/rm_reward_func/mean": -1.598388671875, "rewards/rm_reward_func/std": 19.015661239624023, "step": 338 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 259.65625, "completions/mean_terminated_length": 251.51612854003906, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "epoch": 0.2712, "grad_norm": 26.45867919921875, "kl": 0.097412109375, "learning_rate": 1e-06, "loss": 0.1121, "num_tokens": 4309008.0, "reward": 1.324951171875, "reward_std": 5.575651168823242, "rewards/rm_reward_func/mean": 1.324951171875, "rewards/rm_reward_func/std": 11.64552116394043, "step": 339 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 341.15625, "completions/mean_terminated_length": 284.2083435058594, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "epoch": 0.272, "grad_norm": 4.051329135894775, "kl": 0.10674476623535156, "learning_rate": 1e-06, "loss": -0.0366, "num_tokens": 4325749.0, "reward": 12.0419921875, "reward_std": 3.590223550796509, "rewards/rm_reward_func/mean": 12.0419921875, "rewards/rm_reward_func/std": 17.978811264038086, "step": 340 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 418.0, "completions/mean_length": 245.375, "completions/mean_terminated_length": 217.79310607910156, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "epoch": 0.2728, "grad_norm": 8.272236824035645, "kl": 0.07733154296875, "learning_rate": 1e-06, "loss": 0.3771, "num_tokens": 4337033.0, "reward": 0.50048828125, "reward_std": 5.649558067321777, "rewards/rm_reward_func/mean": 0.50048828125, "rewards/rm_reward_func/std": 5.990384101867676, "step": 341 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 333.78125, "completions/mean_terminated_length": 264.0434875488281, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "epoch": 0.2736, "grad_norm": 3.4469985961914062, "kl": 0.06951904296875, "learning_rate": 1e-06, "loss": 0.0304, "num_tokens": 4349922.0, "reward": 3.2762298583984375, "reward_std": 8.944576263427734, "rewards/rm_reward_func/mean": 3.2762298583984375, "rewards/rm_reward_func/std": 14.431233406066895, "step": 342 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 140.0, "completions/max_terminated_length": 140.0, "completions/mean_length": 68.78125, "completions/mean_terminated_length": 68.78125, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "epoch": 0.2744, "grad_norm": 12.23254680633545, "kl": 0.1826171875, "learning_rate": 1e-06, "loss": 0.1112, "num_tokens": 4355779.0, "reward": -2.1884765625, "reward_std": 3.757051944732666, "rewards/rm_reward_func/mean": -2.1884765625, "rewards/rm_reward_func/std": 9.511407852172852, "step": 343 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 492.0, "completions/mean_length": 284.78125, "completions/mean_terminated_length": 232.34616088867188, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "epoch": 0.2752, "grad_norm": 5.746858596801758, "kl": 0.10980224609375, "learning_rate": 1e-06, "loss": -0.1269, "num_tokens": 4368244.0, "reward": 0.2880859375, "reward_std": 3.1070127487182617, "rewards/rm_reward_func/mean": 0.2880859375, "rewards/rm_reward_func/std": 14.436668395996094, "step": 344 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 388.0, "completions/max_terminated_length": 388.0, "completions/mean_length": 158.59375, "completions/mean_terminated_length": 158.59375, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "epoch": 0.276, "grad_norm": 6.810152053833008, "kl": 0.12841796875, "learning_rate": 1e-06, "loss": -0.1169, "num_tokens": 4378415.0, "reward": 2.64385986328125, "reward_std": 4.024966716766357, "rewards/rm_reward_func/mean": 2.64385986328125, "rewards/rm_reward_func/std": 8.710803985595703, "step": 345 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 486.0, "completions/mean_length": 300.15625, "completions/mean_terminated_length": 189.1904754638672, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "epoch": 0.2768, "grad_norm": 4.056140422821045, "kl": 0.13568115234375, "learning_rate": 1e-06, "loss": 0.0213, "num_tokens": 4390156.0, "reward": 8.573406219482422, "reward_std": 4.77297306060791, "rewards/rm_reward_func/mean": 8.573406219482422, "rewards/rm_reward_func/std": 5.832368850708008, "step": 346 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 387.3125, "completions/mean_terminated_length": 345.75, "completions/min_length": 182.0, "completions/min_terminated_length": 182.0, "epoch": 0.2776, "grad_norm": 3.6153457164764404, "kl": 0.0836181640625, "learning_rate": 1e-06, "loss": -0.0447, "num_tokens": 4405446.0, "reward": 7.1898193359375, "reward_std": 6.768939971923828, "rewards/rm_reward_func/mean": 7.1898193359375, "rewards/rm_reward_func/std": 12.574838638305664, "step": 347 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 436.0, "completions/mean_length": 319.53125, "completions/mean_terminated_length": 204.0500030517578, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "epoch": 0.2784, "grad_norm": 38.247314453125, "kl": 0.071136474609375, "learning_rate": 1e-06, "loss": -0.0091, "num_tokens": 4419767.0, "reward": -6.6529541015625, "reward_std": 5.824492454528809, "rewards/rm_reward_func/mean": -6.6529541015625, "rewards/rm_reward_func/std": 8.597187042236328, "step": 348 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 477.0, "completions/mean_length": 318.875, "completions/mean_terminated_length": 298.89654541015625, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "epoch": 0.2792, "grad_norm": 4.058663845062256, "kl": 0.082733154296875, "learning_rate": 1e-06, "loss": -0.0386, "num_tokens": 4432003.0, "reward": -0.1033935546875, "reward_std": 8.34597396850586, "rewards/rm_reward_func/mean": -0.1033935546875, "rewards/rm_reward_func/std": 15.697948455810547, "step": 349 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 368.59375, "completions/mean_terminated_length": 282.5500183105469, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "epoch": 0.28, "grad_norm": 27.752031326293945, "kl": 0.07891845703125, "learning_rate": 1e-06, "loss": -0.047, "num_tokens": 4449870.0, "reward": 0.47021484375, "reward_std": 6.845122337341309, "rewards/rm_reward_func/mean": 0.47021484375, "rewards/rm_reward_func/std": 11.682084083557129, "step": 350 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 454.0, "completions/mean_length": 242.75, "completions/mean_terminated_length": 234.06451416015625, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "epoch": 0.2808, "grad_norm": 4.997708797454834, "kl": 0.1334228515625, "learning_rate": 1e-06, "loss": -0.0635, "num_tokens": 4462158.0, "reward": 7.95166015625, "reward_std": 8.451520919799805, "rewards/rm_reward_func/mean": 7.95166015625, "rewards/rm_reward_func/std": 13.134011268615723, "step": 351 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 266.6875, "completions/mean_terminated_length": 170.69564819335938, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "epoch": 0.2816, "grad_norm": 11.950886726379395, "kl": 0.12762451171875, "learning_rate": 1e-06, "loss": 0.0629, "num_tokens": 4473372.0, "reward": 2.12890625, "reward_std": 10.61873722076416, "rewards/rm_reward_func/mean": 2.12890625, "rewards/rm_reward_func/std": 17.77381706237793, "step": 352 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 315.84375, "completions/mean_terminated_length": 260.91998291015625, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "epoch": 0.2824, "grad_norm": 9.539039611816406, "kl": 0.135009765625, "learning_rate": 1e-06, "loss": 0.2418, "num_tokens": 4486855.0, "reward": 10.563232421875, "reward_std": 8.17442512512207, "rewards/rm_reward_func/mean": 10.563232421875, "rewards/rm_reward_func/std": 15.37407398223877, "step": 353 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 388.25, "completions/mean_terminated_length": 353.6000061035156, "completions/min_length": 133.0, "completions/min_terminated_length": 133.0, "epoch": 0.2832, "grad_norm": 3.476372003555298, "kl": 0.1109619140625, "learning_rate": 1e-06, "loss": -0.0289, "num_tokens": 4501519.0, "reward": 19.090087890625, "reward_std": 6.380127906799316, "rewards/rm_reward_func/mean": 19.090087890625, "rewards/rm_reward_func/std": 11.411051750183105, "step": 354 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 316.96875, "completions/mean_terminated_length": 262.3599853515625, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "epoch": 0.284, "grad_norm": 5.603827953338623, "kl": 0.129150390625, "learning_rate": 1e-06, "loss": -0.0998, "num_tokens": 4514398.0, "reward": -2.05572509765625, "reward_std": 5.069465637207031, "rewards/rm_reward_func/mean": -2.05572509765625, "rewards/rm_reward_func/std": 7.577589988708496, "step": 355 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 369.90625, "completions/mean_terminated_length": 330.1199951171875, "completions/min_length": 131.0, "completions/min_terminated_length": 131.0, "epoch": 0.2848, "grad_norm": 3.7223684787750244, "kl": 0.1007080078125, "learning_rate": 1e-06, "loss": 0.011, "num_tokens": 4528347.0, "reward": 1.4289093017578125, "reward_std": 5.715549468994141, "rewards/rm_reward_func/mean": 1.4289093017578125, "rewards/rm_reward_func/std": 14.303176879882812, "step": 356 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 335.28125, "completions/mean_terminated_length": 323.5000305175781, "completions/min_length": 145.0, "completions/min_terminated_length": 145.0, "epoch": 0.2856, "grad_norm": 4.0376200675964355, "kl": 0.0750732421875, "learning_rate": 1e-06, "loss": -0.0575, "num_tokens": 4541772.0, "reward": 6.256103515625, "reward_std": 8.777978897094727, "rewards/rm_reward_func/mean": 6.256103515625, "rewards/rm_reward_func/std": 14.211292266845703, "step": 357 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 278.75, "completions/mean_terminated_length": 235.55555725097656, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "epoch": 0.2864, "grad_norm": 4.349524021148682, "kl": 0.197265625, "learning_rate": 1e-06, "loss": 0.0664, "num_tokens": 4552908.0, "reward": 5.4974365234375, "reward_std": 5.0523176193237305, "rewards/rm_reward_func/mean": 5.4974365234375, "rewards/rm_reward_func/std": 15.988303184509277, "step": 358 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 484.0, "completions/mean_length": 248.75, "completions/mean_terminated_length": 240.258056640625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "epoch": 0.2872, "grad_norm": 4.672142028808594, "kl": 0.200927734375, "learning_rate": 1e-06, "loss": 0.0244, "num_tokens": 4563372.0, "reward": -1.47747802734375, "reward_std": 4.878537654876709, "rewards/rm_reward_func/mean": -1.47747802734375, "rewards/rm_reward_func/std": 9.86972713470459, "step": 359 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 277.15625, "completions/mean_terminated_length": 211.39999389648438, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "epoch": 0.288, "grad_norm": 6.796772480010986, "kl": 0.0882568359375, "learning_rate": 1e-06, "loss": -0.0351, "num_tokens": 4574377.0, "reward": 1.7371826171875, "reward_std": 6.135521411895752, "rewards/rm_reward_func/mean": 1.7371826171875, "rewards/rm_reward_func/std": 14.885226249694824, "step": 360 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 493.0, "completions/mean_length": 347.6875, "completions/mean_terminated_length": 336.73333740234375, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "epoch": 0.2888, "grad_norm": 7.4351806640625, "kl": 0.1314697265625, "learning_rate": 1e-06, "loss": -0.1156, "num_tokens": 4587791.0, "reward": -0.7554969787597656, "reward_std": 5.259791851043701, "rewards/rm_reward_func/mean": -0.7554969787597656, "rewards/rm_reward_func/std": 12.53715991973877, "step": 361 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 321.59375, "completions/mean_terminated_length": 294.39288330078125, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "epoch": 0.2896, "grad_norm": 5.127910614013672, "kl": 0.08160400390625, "learning_rate": 1e-06, "loss": -0.0329, "num_tokens": 4600466.0, "reward": 1.8590850830078125, "reward_std": 7.74429988861084, "rewards/rm_reward_func/mean": 1.8590850830078125, "rewards/rm_reward_func/std": 14.955947875976562, "step": 362 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 460.0, "completions/mean_length": 169.59375, "completions/mean_terminated_length": 158.5483856201172, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "epoch": 0.2904, "grad_norm": 4.208690643310547, "kl": 0.1861572265625, "learning_rate": 1e-06, "loss": 0.0933, "num_tokens": 4609349.0, "reward": 3.2529525756835938, "reward_std": 2.592770576477051, "rewards/rm_reward_func/mean": 3.2529525756835938, "rewards/rm_reward_func/std": 9.945049285888672, "step": 363 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 419.96875, "completions/mean_terminated_length": 364.75, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "epoch": 0.2912, "grad_norm": 3.5521671772003174, "kl": 0.079864501953125, "learning_rate": 1e-06, "loss": -0.1582, "num_tokens": 4626428.0, "reward": 2.2109375, "reward_std": 9.090686798095703, "rewards/rm_reward_func/mean": 2.2109375, "rewards/rm_reward_func/std": 24.739288330078125, "step": 364 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 237.53125, "completions/mean_terminated_length": 209.13792419433594, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "epoch": 0.292, "grad_norm": 6.295807838439941, "kl": 0.051605224609375, "learning_rate": 1e-06, "loss": 0.0554, "num_tokens": 4639565.0, "reward": -10.0103759765625, "reward_std": 5.414639472961426, "rewards/rm_reward_func/mean": -10.0103759765625, "rewards/rm_reward_func/std": 11.643550872802734, "step": 365 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 335.84375, "completions/mean_terminated_length": 286.5199890136719, "completions/min_length": 104.0, "completions/min_terminated_length": 104.0, "epoch": 0.2928, "grad_norm": 4.325115203857422, "kl": 0.0908203125, "learning_rate": 1e-06, "loss": -0.0157, "num_tokens": 4652608.0, "reward": 13.1593017578125, "reward_std": 7.079737663269043, "rewards/rm_reward_func/mean": 13.1593017578125, "rewards/rm_reward_func/std": 17.48756980895996, "step": 366 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 364.25, "completions/mean_terminated_length": 275.6000061035156, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "epoch": 0.2936, "grad_norm": 4.418249130249023, "kl": 0.169189453125, "learning_rate": 1e-06, "loss": 0.1931, "num_tokens": 4667984.0, "reward": 5.2593994140625, "reward_std": 7.445150375366211, "rewards/rm_reward_func/mean": 5.2593994140625, "rewards/rm_reward_func/std": 19.743183135986328, "step": 367 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 297.8125, "completions/mean_terminated_length": 258.1481628417969, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "epoch": 0.2944, "grad_norm": 3.9477925300598145, "kl": 0.167724609375, "learning_rate": 1e-06, "loss": -0.0239, "num_tokens": 4681170.0, "reward": 7.892822265625, "reward_std": 4.728097915649414, "rewards/rm_reward_func/mean": 7.892822265625, "rewards/rm_reward_func/std": 5.967565536499023, "step": 368 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 448.0, "completions/mean_length": 281.28125, "completions/mean_terminated_length": 238.55555725097656, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "epoch": 0.2952, "grad_norm": 6.2598443031311035, "kl": 0.108642578125, "learning_rate": 1e-06, "loss": -0.2548, "num_tokens": 4692299.0, "reward": -7.61279296875, "reward_std": 4.482115745544434, "rewards/rm_reward_func/mean": -7.61279296875, "rewards/rm_reward_func/std": 9.240117073059082, "step": 369 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 442.0, "completions/mean_length": 204.84375, "completions/mean_terminated_length": 194.9354705810547, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "epoch": 0.296, "grad_norm": 4.286995887756348, "kl": 0.1629638671875, "learning_rate": 1e-06, "loss": -0.0786, "num_tokens": 4701822.0, "reward": 9.278564453125, "reward_std": 7.017640590667725, "rewards/rm_reward_func/mean": 9.278564453125, "rewards/rm_reward_func/std": 11.16659164428711, "step": 370 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 396.0, "completions/mean_length": 137.71875, "completions/mean_terminated_length": 125.64515686035156, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "epoch": 0.2968, "grad_norm": 6.7198004722595215, "kl": 0.17926025390625, "learning_rate": 1e-06, "loss": 0.0636, "num_tokens": 4709437.0, "reward": 1.2930221557617188, "reward_std": 2.7805871963500977, "rewards/rm_reward_func/mean": 1.2930221557617188, "rewards/rm_reward_func/std": 6.2812652587890625, "step": 371 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 127.0, "completions/max_terminated_length": 127.0, "completions/mean_length": 67.65625, "completions/mean_terminated_length": 67.65625, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "epoch": 0.2976, "grad_norm": 7.943471908569336, "kl": 0.2227783203125, "learning_rate": 1e-06, "loss": -0.0281, "num_tokens": 4716018.0, "reward": 0.7178955078125, "reward_std": 2.993560314178467, "rewards/rm_reward_func/mean": 0.7178955078125, "rewards/rm_reward_func/std": 7.339946746826172, "step": 372 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 326.53125, "completions/mean_terminated_length": 314.16668701171875, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "epoch": 0.2984, "grad_norm": 2.898266315460205, "kl": 0.1676025390625, "learning_rate": 1e-06, "loss": 0.0155, "num_tokens": 4729731.0, "reward": 8.0377197265625, "reward_std": 4.975411891937256, "rewards/rm_reward_func/mean": 8.0377197265625, "rewards/rm_reward_func/std": 11.102714538574219, "step": 373 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 314.59375, "completions/mean_terminated_length": 224.8636474609375, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "epoch": 0.2992, "grad_norm": 5.358304023742676, "kl": 0.1610107421875, "learning_rate": 1e-06, "loss": 0.197, "num_tokens": 4742030.0, "reward": 5.0345458984375, "reward_std": 6.491810321807861, "rewards/rm_reward_func/mean": 5.0345458984375, "rewards/rm_reward_func/std": 8.370210647583008, "step": 374 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.46875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 426.375, "completions/mean_terminated_length": 350.8235168457031, "completions/min_length": 92.0, "completions/min_terminated_length": 92.0, "epoch": 0.3, "grad_norm": 4.6803483963012695, "kl": 0.0802001953125, "learning_rate": 1e-06, "loss": -0.0817, "num_tokens": 4761410.0, "reward": 10.879684448242188, "reward_std": 8.08679485321045, "rewards/rm_reward_func/mean": 10.879684448242188, "rewards/rm_reward_func/std": 10.522982597351074, "step": 375 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 342.0, "completions/mean_length": 204.09375, "completions/mean_terminated_length": 194.16128540039062, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "epoch": 0.3008, "grad_norm": 3.7969260215759277, "kl": 0.1573486328125, "learning_rate": 1e-06, "loss": -0.011, "num_tokens": 4770525.0, "reward": 9.822998046875, "reward_std": 6.23846435546875, "rewards/rm_reward_func/mean": 9.822998046875, "rewards/rm_reward_func/std": 9.928614616394043, "step": 376 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 231.34375, "completions/mean_terminated_length": 212.6333465576172, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "epoch": 0.3016, "grad_norm": 4.173020839691162, "kl": 0.240234375, "learning_rate": 1e-06, "loss": -0.0314, "num_tokens": 4780584.0, "reward": 7.1978607177734375, "reward_std": 5.048998832702637, "rewards/rm_reward_func/mean": 7.1978607177734375, "rewards/rm_reward_func/std": 9.145500183105469, "step": 377 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 312.96875, "completions/mean_terminated_length": 246.625, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "epoch": 0.3024, "grad_norm": 3.5591626167297363, "kl": 0.175537109375, "learning_rate": 1e-06, "loss": 0.0296, "num_tokens": 4792983.0, "reward": 1.46685791015625, "reward_std": 3.8416523933410645, "rewards/rm_reward_func/mean": 1.46685791015625, "rewards/rm_reward_func/std": 8.133792877197266, "step": 378 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 497.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 240.4375, "completions/mean_terminated_length": 240.4375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "epoch": 0.3032, "grad_norm": 4.513542652130127, "kl": 0.148193359375, "learning_rate": 1e-06, "loss": -0.0096, "num_tokens": 4803181.0, "reward": 3.9730072021484375, "reward_std": 3.2463040351867676, "rewards/rm_reward_func/mean": 3.9730072021484375, "rewards/rm_reward_func/std": 7.680938243865967, "step": 379 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 274.78125, "completions/mean_terminated_length": 250.2413787841797, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "epoch": 0.304, "grad_norm": 4.1872639656066895, "kl": 0.088134765625, "learning_rate": 1e-06, "loss": 0.2139, "num_tokens": 4816062.0, "reward": 8.164306640625, "reward_std": 9.427093505859375, "rewards/rm_reward_func/mean": 8.164306640625, "rewards/rm_reward_func/std": 11.66413402557373, "step": 380 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 318.28125, "completions/mean_terminated_length": 242.478271484375, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "epoch": 0.3048, "grad_norm": 4.635010242462158, "kl": 0.1771240234375, "learning_rate": 1e-06, "loss": -0.0366, "num_tokens": 4828711.0, "reward": 7.5330810546875, "reward_std": 4.666987419128418, "rewards/rm_reward_func/mean": 7.5330810546875, "rewards/rm_reward_func/std": 6.908311367034912, "step": 381 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 274.90625, "completions/mean_terminated_length": 220.19232177734375, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "epoch": 0.3056, "grad_norm": 4.149857044219971, "kl": 0.168212890625, "learning_rate": 1e-06, "loss": 0.0295, "num_tokens": 4839428.0, "reward": -3.975341796875, "reward_std": 4.726168632507324, "rewards/rm_reward_func/mean": -3.975341796875, "rewards/rm_reward_func/std": 9.026304244995117, "step": 382 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 276.0, "completions/mean_terminated_length": 209.9199981689453, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "epoch": 0.3064, "grad_norm": 3.875955820083618, "kl": 0.13720703125, "learning_rate": 1e-06, "loss": 0.0275, "num_tokens": 4850612.0, "reward": -5.6435546875, "reward_std": 3.571234941482544, "rewards/rm_reward_func/mean": -5.6435546875, "rewards/rm_reward_func/std": 16.429338455200195, "step": 383 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 227.34375, "completions/mean_terminated_length": 208.36668395996094, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "epoch": 0.3072, "grad_norm": 3.766765832901001, "kl": 0.154052734375, "learning_rate": 1e-06, "loss": 0.1151, "num_tokens": 4862007.0, "reward": 9.3204345703125, "reward_std": 5.31584358215332, "rewards/rm_reward_func/mean": 9.3204345703125, "rewards/rm_reward_func/std": 12.34428596496582, "step": 384 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 274.90625, "completions/mean_terminated_length": 220.19232177734375, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "epoch": 0.308, "grad_norm": 3.00742506980896, "kl": 0.134521484375, "learning_rate": 1e-06, "loss": -0.0583, "num_tokens": 4873196.0, "reward": -3.42340087890625, "reward_std": 3.529475450515747, "rewards/rm_reward_func/mean": -3.42340087890625, "rewards/rm_reward_func/std": 14.037924766540527, "step": 385 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 375.53125, "completions/mean_terminated_length": 293.6499938964844, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "epoch": 0.3088, "grad_norm": 4.225069046020508, "kl": 0.15325927734375, "learning_rate": 1e-06, "loss": -0.1237, "num_tokens": 4889389.0, "reward": 10.78369140625, "reward_std": 9.285670280456543, "rewards/rm_reward_func/mean": 10.78369140625, "rewards/rm_reward_func/std": 19.006528854370117, "step": 386 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 340.40625, "completions/mean_terminated_length": 262.4090881347656, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "epoch": 0.3096, "grad_norm": 2.9847397804260254, "kl": 0.14892578125, "learning_rate": 1e-06, "loss": 0.0144, "num_tokens": 4903578.0, "reward": 9.98577880859375, "reward_std": 4.354106426239014, "rewards/rm_reward_func/mean": 9.98577880859375, "rewards/rm_reward_func/std": 9.918432235717773, "step": 387 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 282.96875, "completions/mean_terminated_length": 259.2758483886719, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "epoch": 0.3104, "grad_norm": 3.2981696128845215, "kl": 0.138427734375, "learning_rate": 1e-06, "loss": 0.0006, "num_tokens": 4915585.0, "reward": 6.951171875, "reward_std": 4.37175178527832, "rewards/rm_reward_func/mean": 6.951171875, "rewards/rm_reward_func/std": 16.657833099365234, "step": 388 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 282.0625, "completions/mean_terminated_length": 258.2758483886719, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "epoch": 0.3112, "grad_norm": 4.533021450042725, "kl": 0.1182861328125, "learning_rate": 1e-06, "loss": -0.1073, "num_tokens": 4929283.0, "reward": 1.020263671875, "reward_std": 5.0986127853393555, "rewards/rm_reward_func/mean": 1.020263671875, "rewards/rm_reward_func/std": 10.977892875671387, "step": 389 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 252.9375, "completions/mean_terminated_length": 193.1538543701172, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "epoch": 0.312, "grad_norm": 4.486575603485107, "kl": 0.15936279296875, "learning_rate": 1e-06, "loss": -0.0062, "num_tokens": 4941017.0, "reward": -1.0895843505859375, "reward_std": 5.964107990264893, "rewards/rm_reward_func/mean": -1.0895843505859375, "rewards/rm_reward_func/std": 11.993568420410156, "step": 390 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 404.875, "completions/mean_terminated_length": 380.15386962890625, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.3128, "grad_norm": 3.2242486476898193, "kl": 0.097900390625, "learning_rate": 1e-06, "loss": 0.0056, "num_tokens": 4955989.0, "reward": 17.81597900390625, "reward_std": 8.450302124023438, "rewards/rm_reward_func/mean": 17.81597900390625, "rewards/rm_reward_func/std": 14.226698875427246, "step": 391 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 275.53125, "completions/mean_terminated_length": 220.9615478515625, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "epoch": 0.3136, "grad_norm": 3.7851226329803467, "kl": 0.231781005859375, "learning_rate": 1e-06, "loss": 0.0319, "num_tokens": 4967710.0, "reward": 13.55126953125, "reward_std": 4.914253234863281, "rewards/rm_reward_func/mean": 13.55126953125, "rewards/rm_reward_func/std": 15.68505859375, "step": 392 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 479.0, "completions/mean_length": 208.75, "completions/mean_terminated_length": 188.53334045410156, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "epoch": 0.3144, "grad_norm": 3.546250104904175, "kl": 0.248779296875, "learning_rate": 1e-06, "loss": 0.1134, "num_tokens": 4977470.0, "reward": 6.796875, "reward_std": 4.045927047729492, "rewards/rm_reward_func/mean": 6.796875, "rewards/rm_reward_func/std": 9.776312828063965, "step": 393 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 285.1875, "completions/mean_terminated_length": 252.7857208251953, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "epoch": 0.3152, "grad_norm": 8.546547889709473, "kl": 0.190673828125, "learning_rate": 1e-06, "loss": 0.2501, "num_tokens": 4990396.0, "reward": 12.95703125, "reward_std": 6.034456729888916, "rewards/rm_reward_func/mean": 12.95703125, "rewards/rm_reward_func/std": 11.617748260498047, "step": 394 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 449.0, "completions/max_terminated_length": 449.0, "completions/mean_length": 294.84375, "completions/mean_terminated_length": 294.84375, "completions/min_length": 125.0, "completions/min_terminated_length": 125.0, "epoch": 0.316, "grad_norm": 3.5022923946380615, "kl": 0.162109375, "learning_rate": 1e-06, "loss": -0.0858, "num_tokens": 5002815.0, "reward": 14.046142578125, "reward_std": 6.6521525382995605, "rewards/rm_reward_func/mean": 14.046142578125, "rewards/rm_reward_func/std": 15.969138145446777, "step": 395 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40625, "completions/max_length": 512.0, "completions/max_terminated_length": 492.0, "completions/mean_length": 406.75, "completions/mean_terminated_length": 334.7368469238281, "completions/min_length": 179.0, "completions/min_terminated_length": 179.0, "epoch": 0.3168, "grad_norm": 3.3227157592773438, "kl": 0.09228515625, "learning_rate": 1e-06, "loss": -0.0137, "num_tokens": 5021671.0, "reward": 3.28466796875, "reward_std": 4.573013782501221, "rewards/rm_reward_func/mean": 3.28466796875, "rewards/rm_reward_func/std": 10.546538352966309, "step": 396 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 480.0, "completions/max_terminated_length": 480.0, "completions/mean_length": 199.3125, "completions/mean_terminated_length": 199.3125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "epoch": 0.3176, "grad_norm": 8.376974105834961, "kl": 0.1419677734375, "learning_rate": 1e-06, "loss": 0.01, "num_tokens": 5033201.0, "reward": 3.65576171875, "reward_std": 4.244373321533203, "rewards/rm_reward_func/mean": 3.65576171875, "rewards/rm_reward_func/std": 15.917009353637695, "step": 397 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 478.0, "completions/max_terminated_length": 478.0, "completions/mean_length": 303.9375, "completions/mean_terminated_length": 303.9375, "completions/min_length": 110.0, "completions/min_terminated_length": 110.0, "epoch": 0.3184, "grad_norm": 3.5459303855895996, "kl": 0.10992431640625, "learning_rate": 1e-06, "loss": 0.0089, "num_tokens": 5044935.0, "reward": 5.091552734375, "reward_std": 4.916255950927734, "rewards/rm_reward_func/mean": 5.091552734375, "rewards/rm_reward_func/std": 14.6499605178833, "step": 398 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 317.90625, "completions/mean_terminated_length": 281.96295166015625, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "epoch": 0.3192, "grad_norm": 3.4768733978271484, "kl": 0.1575927734375, "learning_rate": 1e-06, "loss": 0.0603, "num_tokens": 5058220.0, "reward": 2.6343994140625, "reward_std": 6.248598098754883, "rewards/rm_reward_func/mean": 2.6343994140625, "rewards/rm_reward_func/std": 12.041556358337402, "step": 399 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 495.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 312.09375, "completions/mean_terminated_length": 312.09375, "completions/min_length": 106.0, "completions/min_terminated_length": 106.0, "epoch": 0.32, "grad_norm": 3.6274449825286865, "kl": 0.13006591796875, "learning_rate": 1e-06, "loss": 0.0541, "num_tokens": 5070567.0, "reward": 6.393798828125, "reward_std": 4.916718006134033, "rewards/rm_reward_func/mean": 6.393798828125, "rewards/rm_reward_func/std": 14.114630699157715, "step": 400 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 120.0, "completions/max_terminated_length": 120.0, "completions/mean_length": 77.9375, "completions/mean_terminated_length": 77.9375, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "epoch": 0.3208, "grad_norm": 4.715888023376465, "kl": 0.33203125, "learning_rate": 1e-06, "loss": -0.0051, "num_tokens": 5077253.0, "reward": 2.888671875, "reward_std": 3.143357038497925, "rewards/rm_reward_func/mean": 2.888671875, "rewards/rm_reward_func/std": 10.683290481567383, "step": 401 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 297.0, "completions/mean_length": 140.5625, "completions/mean_terminated_length": 115.80000305175781, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "epoch": 0.3216, "grad_norm": 7.256206512451172, "kl": 0.289306640625, "learning_rate": 1e-06, "loss": 0.432, "num_tokens": 5085031.0, "reward": 2.5806884765625, "reward_std": 4.735476016998291, "rewards/rm_reward_func/mean": 2.5806884765625, "rewards/rm_reward_func/std": 8.712172508239746, "step": 402 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 465.0, "completions/mean_length": 286.125, "completions/mean_terminated_length": 262.75860595703125, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "epoch": 0.3224, "grad_norm": 6.090017795562744, "kl": 0.133056640625, "learning_rate": 1e-06, "loss": -0.2161, "num_tokens": 5097227.0, "reward": 3.853515625, "reward_std": 5.223775863647461, "rewards/rm_reward_func/mean": 3.853515625, "rewards/rm_reward_func/std": 14.418701171875, "step": 403 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 303.3125, "completions/mean_terminated_length": 281.7241516113281, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "epoch": 0.3232, "grad_norm": 4.603776931762695, "kl": 0.0992431640625, "learning_rate": 1e-06, "loss": 0.0173, "num_tokens": 5109629.0, "reward": 6.81072998046875, "reward_std": 10.010828018188477, "rewards/rm_reward_func/mean": 6.81072998046875, "rewards/rm_reward_func/std": 19.7315673828125, "step": 404 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 327.40625, "completions/mean_terminated_length": 265.875, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "epoch": 0.324, "grad_norm": 3.5028727054595947, "kl": 0.10601806640625, "learning_rate": 1e-06, "loss": -0.067, "num_tokens": 5122570.0, "reward": -3.7301025390625, "reward_std": 4.875129222869873, "rewards/rm_reward_func/mean": -3.7301025390625, "rewards/rm_reward_func/std": 7.149316787719727, "step": 405 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 360.21875, "completions/mean_terminated_length": 280.71429443359375, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "epoch": 0.3248, "grad_norm": 5.7773871421813965, "kl": 0.1090087890625, "learning_rate": 1e-06, "loss": -0.3303, "num_tokens": 5136473.0, "reward": 0.67950439453125, "reward_std": 9.863033294677734, "rewards/rm_reward_func/mean": 0.67950439453125, "rewards/rm_reward_func/std": 14.524174690246582, "step": 406 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 461.0, "completions/mean_length": 238.84375, "completions/mean_terminated_length": 162.36000061035156, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "epoch": 0.3256, "grad_norm": 2.8504834175109863, "kl": 0.235107421875, "learning_rate": 1e-06, "loss": -0.0181, "num_tokens": 5147908.0, "reward": 5.07373046875, "reward_std": 3.009080648422241, "rewards/rm_reward_func/mean": 5.07373046875, "rewards/rm_reward_func/std": 7.276069641113281, "step": 407 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 342.375, "completions/mean_terminated_length": 285.8333435058594, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "epoch": 0.3264, "grad_norm": 6.275087833404541, "kl": 0.151123046875, "learning_rate": 1e-06, "loss": 0.237, "num_tokens": 5161120.0, "reward": 6.7158203125, "reward_std": 6.726596832275391, "rewards/rm_reward_func/mean": 6.7158203125, "rewards/rm_reward_func/std": 7.3756632804870605, "step": 408 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 417.15625, "completions/mean_terminated_length": 390.6000061035156, "completions/min_length": 235.0, "completions/min_terminated_length": 235.0, "epoch": 0.3272, "grad_norm": 3.3832011222839355, "kl": 0.0989990234375, "learning_rate": 1e-06, "loss": 0.0316, "num_tokens": 5178797.0, "reward": 6.41094970703125, "reward_std": 5.119528293609619, "rewards/rm_reward_func/mean": 6.41094970703125, "rewards/rm_reward_func/std": 7.976118564605713, "step": 409 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 425.0, "completions/mean_length": 293.25, "completions/mean_terminated_length": 286.19354248046875, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "epoch": 0.328, "grad_norm": 3.2517099380493164, "kl": 0.17236328125, "learning_rate": 1e-06, "loss": 0.0392, "num_tokens": 5190149.0, "reward": 9.473388671875, "reward_std": 6.672955513000488, "rewards/rm_reward_func/mean": 9.473388671875, "rewards/rm_reward_func/std": 11.397151947021484, "step": 410 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 407.0, "completions/mean_length": 304.125, "completions/mean_terminated_length": 209.63636779785156, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "epoch": 0.3288, "grad_norm": 6.965356349945068, "kl": 0.15704345703125, "learning_rate": 1e-06, "loss": 0.2693, "num_tokens": 5203601.0, "reward": -2.846466064453125, "reward_std": 3.277106761932373, "rewards/rm_reward_func/mean": -2.846466064453125, "rewards/rm_reward_func/std": 8.159171104431152, "step": 411 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 452.0, "completions/mean_length": 329.3125, "completions/mean_terminated_length": 268.41668701171875, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "epoch": 0.3296, "grad_norm": 3.6799185276031494, "kl": 0.079345703125, "learning_rate": 1e-06, "loss": -0.0095, "num_tokens": 5217035.0, "reward": -4.873779296875, "reward_std": 5.421168327331543, "rewards/rm_reward_func/mean": -4.873779296875, "rewards/rm_reward_func/std": 10.083694458007812, "step": 412 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 466.0, "completions/mean_length": 382.3125, "completions/mean_terminated_length": 358.2962951660156, "completions/min_length": 149.0, "completions/min_terminated_length": 149.0, "epoch": 0.3304, "grad_norm": 3.278068780899048, "kl": 0.1441650390625, "learning_rate": 1e-06, "loss": 0.0341, "num_tokens": 5232501.0, "reward": 10.257858276367188, "reward_std": 8.294170379638672, "rewards/rm_reward_func/mean": 10.257858276367188, "rewards/rm_reward_func/std": 13.173551559448242, "step": 413 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 482.0, "completions/mean_length": 304.84375, "completions/mean_terminated_length": 235.7916717529297, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "epoch": 0.3312, "grad_norm": 3.0338048934936523, "kl": 0.15618896484375, "learning_rate": 1e-06, "loss": -0.1137, "num_tokens": 5244808.0, "reward": 5.3134765625, "reward_std": 6.2857160568237305, "rewards/rm_reward_func/mean": 5.3134765625, "rewards/rm_reward_func/std": 16.0675106048584, "step": 414 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 249.125, "completions/mean_terminated_length": 200.44444274902344, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "epoch": 0.332, "grad_norm": 6.477999687194824, "kl": 0.2847900390625, "learning_rate": 1e-06, "loss": 0.1968, "num_tokens": 5256652.0, "reward": 9.817138671875, "reward_std": 10.444900512695312, "rewards/rm_reward_func/mean": 9.817138671875, "rewards/rm_reward_func/std": 11.969996452331543, "step": 415 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 463.0, "completions/mean_length": 409.21875, "completions/mean_terminated_length": 347.5500183105469, "completions/min_length": 217.0, "completions/min_terminated_length": 217.0, "epoch": 0.3328, "grad_norm": 3.6648788452148438, "kl": 0.2205810546875, "learning_rate": 1e-06, "loss": -0.0097, "num_tokens": 5273531.0, "reward": -2.630218505859375, "reward_std": 3.172177314758301, "rewards/rm_reward_func/mean": -2.630218505859375, "rewards/rm_reward_func/std": 7.456665515899658, "step": 416 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 338.96875, "completions/mean_terminated_length": 314.25, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "epoch": 0.3336, "grad_norm": 4.057036876678467, "kl": 0.1800537109375, "learning_rate": 1e-06, "loss": 0.0165, "num_tokens": 5288122.0, "reward": 7.453369140625, "reward_std": 4.4758758544921875, "rewards/rm_reward_func/mean": 7.453369140625, "rewards/rm_reward_func/std": 8.864148139953613, "step": 417 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 271.8125, "completions/mean_terminated_length": 264.06451416015625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "epoch": 0.3344, "grad_norm": 7.1819305419921875, "kl": 0.20086669921875, "learning_rate": 1e-06, "loss": 0.258, "num_tokens": 5302668.0, "reward": 4.56103515625, "reward_std": 5.861100673675537, "rewards/rm_reward_func/mean": 4.56103515625, "rewards/rm_reward_func/std": 10.66663932800293, "step": 418 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 449.9375, "completions/mean_terminated_length": 421.727294921875, "completions/min_length": 332.0, "completions/min_terminated_length": 332.0, "epoch": 0.3352, "grad_norm": 4.249541282653809, "kl": 0.2069091796875, "learning_rate": 1e-06, "loss": 0.0362, "num_tokens": 5319242.0, "reward": 11.88818359375, "reward_std": 10.299996376037598, "rewards/rm_reward_func/mean": 11.88818359375, "rewards/rm_reward_func/std": 14.682938575744629, "step": 419 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 301.375, "completions/mean_terminated_length": 294.58062744140625, "completions/min_length": 107.0, "completions/min_terminated_length": 107.0, "epoch": 0.336, "grad_norm": 4.146368980407715, "kl": 0.2005615234375, "learning_rate": 1e-06, "loss": 0.1198, "num_tokens": 5334606.0, "reward": 1.5439453125, "reward_std": 5.175732612609863, "rewards/rm_reward_func/mean": 1.5439453125, "rewards/rm_reward_func/std": 11.174506187438965, "step": 420 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 491.0, "completions/mean_length": 240.125, "completions/mean_terminated_length": 222.00001525878906, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "epoch": 0.3368, "grad_norm": 5.178651332855225, "kl": 0.252685546875, "learning_rate": 1e-06, "loss": 0.3182, "num_tokens": 5345146.0, "reward": -5.8485107421875, "reward_std": 10.920194625854492, "rewards/rm_reward_func/mean": -5.8485107421875, "rewards/rm_reward_func/std": 11.926911354064941, "step": 421 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 258.625, "completions/mean_terminated_length": 222.4285888671875, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "epoch": 0.3376, "grad_norm": 7.027132511138916, "kl": 0.424560546875, "learning_rate": 1e-06, "loss": 0.2645, "num_tokens": 5356774.0, "reward": 10.3265380859375, "reward_std": 7.619661331176758, "rewards/rm_reward_func/mean": 10.3265380859375, "rewards/rm_reward_func/std": 16.47081184387207, "step": 422 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 366.71875, "completions/mean_terminated_length": 333.19232177734375, "completions/min_length": 184.0, "completions/min_terminated_length": 184.0, "epoch": 0.3384, "grad_norm": 5.974679470062256, "kl": 0.1844482421875, "learning_rate": 1e-06, "loss": 0.0469, "num_tokens": 5370701.0, "reward": 20.43310546875, "reward_std": 9.503026008605957, "rewards/rm_reward_func/mean": 20.43310546875, "rewards/rm_reward_func/std": 18.206256866455078, "step": 423 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 346.90625, "completions/mean_terminated_length": 300.67999267578125, "completions/min_length": 123.0, "completions/min_terminated_length": 123.0, "epoch": 0.3392, "grad_norm": 4.590559482574463, "kl": 0.110626220703125, "learning_rate": 1e-06, "loss": 0.0688, "num_tokens": 5384434.0, "reward": 1.7897424697875977, "reward_std": 6.459235191345215, "rewards/rm_reward_func/mean": 1.7897424697875977, "rewards/rm_reward_func/std": 15.98327350616455, "step": 424 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.4375, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 423.15625, "completions/mean_terminated_length": 354.0555725097656, "completions/min_length": 82.0, "completions/min_terminated_length": 82.0, "epoch": 0.34, "grad_norm": 16.818695068359375, "kl": 0.2874755859375, "learning_rate": 1e-06, "loss": -0.0521, "num_tokens": 5400303.0, "reward": 3.7025909423828125, "reward_std": 7.56202507019043, "rewards/rm_reward_func/mean": 3.7025909423828125, "rewards/rm_reward_func/std": 20.20536231994629, "step": 425 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 353.71875, "completions/mean_terminated_length": 281.7727355957031, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "epoch": 0.3408, "grad_norm": 5.755769729614258, "kl": 0.47705078125, "learning_rate": 1e-06, "loss": 0.2143, "num_tokens": 5413798.0, "reward": -4.92926025390625, "reward_std": 6.647279739379883, "rewards/rm_reward_func/mean": -4.92926025390625, "rewards/rm_reward_func/std": 9.619757652282715, "step": 426 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40625, "completions/max_length": 512.0, "completions/max_terminated_length": 436.0, "completions/mean_length": 363.4375, "completions/mean_terminated_length": 261.78948974609375, "completions/min_length": 138.0, "completions/min_terminated_length": 138.0, "epoch": 0.3416, "grad_norm": 5.360143661499023, "kl": 0.496826171875, "learning_rate": 1e-06, "loss": 0.1156, "num_tokens": 5427236.0, "reward": -9.527587890625, "reward_std": 7.441409111022949, "rewards/rm_reward_func/mean": -9.527587890625, "rewards/rm_reward_func/std": 8.035903930664062, "step": 427 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 399.0, "completions/mean_length": 301.53125, "completions/mean_terminated_length": 191.2857208251953, "completions/min_length": 23.0, "completions/min_terminated_length": 23.0, "epoch": 0.3424, "grad_norm": 5.262341022491455, "kl": 0.5040283203125, "learning_rate": 1e-06, "loss": 0.1718, "num_tokens": 5441501.0, "reward": -8.2998046875, "reward_std": 4.3596343994140625, "rewards/rm_reward_func/mean": -8.2998046875, "rewards/rm_reward_func/std": 10.656414031982422, "step": 428 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.75, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 498.40625, "completions/mean_terminated_length": 457.625, "completions/min_length": 409.0, "completions/min_terminated_length": 409.0, "epoch": 0.3432, "grad_norm": 5.260901927947998, "kl": 0.877197265625, "learning_rate": 1e-06, "loss": 0.0611, "num_tokens": 5460434.0, "reward": -9.208251953125, "reward_std": 7.153204917907715, "rewards/rm_reward_func/mean": -9.208251953125, "rewards/rm_reward_func/std": 9.89172077178955, "step": 429 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 246.0625, "completions/mean_terminated_length": 208.07144165039062, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "epoch": 0.344, "grad_norm": 7.583652496337891, "kl": 0.421142578125, "learning_rate": 1e-06, "loss": 0.2763, "num_tokens": 5470404.0, "reward": -1.38763427734375, "reward_std": 8.258423805236816, "rewards/rm_reward_func/mean": -1.38763427734375, "rewards/rm_reward_func/std": 14.190618515014648, "step": 430 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.59375, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 375.84375, "completions/mean_terminated_length": 176.84616088867188, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "epoch": 0.3448, "grad_norm": 7.52949857711792, "kl": 1.322265625, "learning_rate": 1e-06, "loss": 0.3466, "num_tokens": 5486727.0, "reward": -10.94189453125, "reward_std": 8.947233200073242, "rewards/rm_reward_func/mean": -10.94189453125, "rewards/rm_reward_func/std": 11.468242645263672, "step": 431 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 512.0, "completions/max_terminated_length": 373.0, "completions/mean_length": 336.09375, "completions/mean_terminated_length": 160.1875, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "epoch": 0.3456, "grad_norm": 6.60636568069458, "kl": 1.8037109375, "learning_rate": 1e-06, "loss": 0.1765, "num_tokens": 5503626.0, "reward": -11.5931396484375, "reward_std": 7.840054035186768, "rewards/rm_reward_func/mean": -11.5931396484375, "rewards/rm_reward_func/std": 13.185185432434082, "step": 432 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.625, "completions/max_length": 512.0, "completions/max_terminated_length": 371.0, "completions/mean_length": 373.71875, "completions/mean_terminated_length": 143.25, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "epoch": 0.3464, "grad_norm": 20.934152603149414, "kl": 1.482421875, "learning_rate": 1e-06, "loss": 0.4161, "num_tokens": 5518233.0, "reward": -11.6781005859375, "reward_std": 9.535715103149414, "rewards/rm_reward_func/mean": -11.6781005859375, "rewards/rm_reward_func/std": 12.062625885009766, "step": 433 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.625, "completions/max_length": 512.0, "completions/max_terminated_length": 415.0, "completions/mean_length": 397.625, "completions/mean_terminated_length": 207.0, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "epoch": 0.3472, "grad_norm": 6.227436542510986, "kl": 1.304443359375, "learning_rate": 1e-06, "loss": 0.1463, "num_tokens": 5535941.0, "reward": -11.416015625, "reward_std": 10.354272842407227, "rewards/rm_reward_func/mean": -11.416015625, "rewards/rm_reward_func/std": 10.621146202087402, "step": 434 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.8125, "completions/max_length": 512.0, "completions/max_terminated_length": 434.0, "completions/mean_length": 450.34375, "completions/mean_terminated_length": 183.1666717529297, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "epoch": 0.348, "grad_norm": 24.91790199279785, "kl": 2.39453125, "learning_rate": 1e-06, "loss": 0.1695, "num_tokens": 5553336.0, "reward": -16.6943359375, "reward_std": 5.182467460632324, "rewards/rm_reward_func/mean": -16.6943359375, "rewards/rm_reward_func/std": 7.9551005363464355, "step": 435 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.84375, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 496.6875, "completions/mean_terminated_length": 414.0, "completions/min_length": 306.0, "completions/min_terminated_length": 306.0, "epoch": 0.3488, "grad_norm": 7.014784812927246, "kl": 2.12548828125, "learning_rate": 1e-06, "loss": 0.1083, "num_tokens": 5571502.0, "reward": -18.14990234375, "reward_std": 7.876721382141113, "rewards/rm_reward_func/mean": -18.14990234375, "rewards/rm_reward_func/std": 9.936715126037598, "step": 436 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.78125, "completions/max_length": 512.0, "completions/max_terminated_length": 453.0, "completions/mean_length": 435.375, "completions/mean_terminated_length": 161.71429443359375, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "epoch": 0.3496, "grad_norm": 6.28053617477417, "kl": 3.0546875, "learning_rate": 1e-06, "loss": 0.3596, "num_tokens": 5588674.0, "reward": -16.22265625, "reward_std": 9.346332550048828, "rewards/rm_reward_func/mean": -16.22265625, "rewards/rm_reward_func/std": 12.737253189086914, "step": 437 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.84375, "completions/max_length": 512.0, "completions/max_terminated_length": 431.0, "completions/mean_length": 487.4375, "completions/mean_terminated_length": 354.8000183105469, "completions/min_length": 267.0, "completions/min_terminated_length": 267.0, "epoch": 0.3504, "grad_norm": 6.686891078948975, "kl": 2.892578125, "learning_rate": 1e-06, "loss": 0.1816, "num_tokens": 5608344.0, "reward": -13.66796875, "reward_std": 10.213824272155762, "rewards/rm_reward_func/mean": -13.66796875, "rewards/rm_reward_func/std": 13.559255599975586, "step": 438 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.65625, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 475.03125, "completions/mean_terminated_length": 404.4545593261719, "completions/min_length": 330.0, "completions/min_terminated_length": 330.0, "epoch": 0.3512, "grad_norm": 4.943912506103516, "kl": 1.7698974609375, "learning_rate": 1e-06, "loss": 0.091, "num_tokens": 5629521.0, "reward": -12.60577392578125, "reward_std": 10.764892578125, "rewards/rm_reward_func/mean": -12.60577392578125, "rewards/rm_reward_func/std": 13.864947319030762, "step": 439 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.6875, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 449.28125, "completions/mean_terminated_length": 311.3000183105469, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "epoch": 0.352, "grad_norm": 10.61517333984375, "kl": 3.8173828125, "learning_rate": 1e-06, "loss": 0.1559, "num_tokens": 5647154.0, "reward": -16.4229736328125, "reward_std": 4.779942989349365, "rewards/rm_reward_func/mean": -16.4229736328125, "rewards/rm_reward_func/std": 6.662553787231445, "step": 440 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.65625, "completions/max_length": 512.0, "completions/max_terminated_length": 459.0, "completions/mean_length": 438.65625, "completions/mean_terminated_length": 298.6363830566406, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "epoch": 0.3528, "grad_norm": 9.699241638183594, "kl": 3.1298828125, "learning_rate": 1e-06, "loss": 0.2842, "num_tokens": 5664383.0, "reward": -12.2291259765625, "reward_std": 8.05883502960205, "rewards/rm_reward_func/mean": -12.2291259765625, "rewards/rm_reward_func/std": 13.489725112915039, "step": 441 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.625, "completions/max_length": 512.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 356.9375, "completions/mean_terminated_length": 98.5, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "epoch": 0.3536, "grad_norm": 8.036911010742188, "kl": 3.30029296875, "learning_rate": 1e-06, "loss": 0.2722, "num_tokens": 5678285.0, "reward": -16.7509765625, "reward_std": 7.531893730163574, "rewards/rm_reward_func/mean": -16.7509765625, "rewards/rm_reward_func/std": 9.410063743591309, "step": 442 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.875, "completions/max_length": 512.0, "completions/max_terminated_length": 484.0, "completions/mean_length": 502.78125, "completions/mean_terminated_length": 438.25, "completions/min_length": 400.0, "completions/min_terminated_length": 400.0, "epoch": 0.3544, "grad_norm": 11.907139778137207, "kl": 4.3046875, "learning_rate": 1e-06, "loss": 0.1862, "num_tokens": 5696630.0, "reward": -15.4208984375, "reward_std": 7.1098833084106445, "rewards/rm_reward_func/mean": -15.4208984375, "rewards/rm_reward_func/std": 17.015018463134766, "step": 443 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.71875, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 423.71875, "completions/mean_terminated_length": 198.11111450195312, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "epoch": 0.3552, "grad_norm": 131.54908752441406, "kl": 3.455078125, "learning_rate": 1e-06, "loss": 0.1955, "num_tokens": 5713213.0, "reward": -18.561279296875, "reward_std": 7.316192626953125, "rewards/rm_reward_func/mean": -18.561279296875, "rewards/rm_reward_func/std": 9.157906532287598, "step": 444 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.78125, "completions/max_length": 512.0, "completions/max_terminated_length": 465.0, "completions/mean_length": 462.34375, "completions/mean_terminated_length": 285.0, "completions/min_length": 135.0, "completions/min_terminated_length": 135.0, "epoch": 0.356, "grad_norm": 5.364108085632324, "kl": 3.9921875, "learning_rate": 1e-06, "loss": 0.2971, "num_tokens": 5729992.0, "reward": -8.744140625, "reward_std": 15.03476333618164, "rewards/rm_reward_func/mean": -8.744140625, "rewards/rm_reward_func/std": 19.18375587463379, "step": 445 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.75, "completions/max_length": 512.0, "completions/max_terminated_length": 476.0, "completions/mean_length": 434.6875, "completions/mean_terminated_length": 202.75, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "epoch": 0.3568, "grad_norm": 8.373677253723145, "kl": 4.99609375, "learning_rate": 1e-06, "loss": 0.2947, "num_tokens": 5747238.0, "reward": -14.09228515625, "reward_std": 10.04998779296875, "rewards/rm_reward_func/mean": -14.09228515625, "rewards/rm_reward_func/std": 12.171612739562988, "step": 446 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.59375, "completions/max_length": 512.0, "completions/max_terminated_length": 463.0, "completions/mean_length": 425.25, "completions/mean_terminated_length": 298.4615478515625, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.3576, "grad_norm": 6.68599271774292, "kl": 3.321533203125, "learning_rate": 1e-06, "loss": 0.3343, "num_tokens": 5764894.0, "reward": -3.4609375, "reward_std": 9.75390625, "rewards/rm_reward_func/mean": -3.4609375, "rewards/rm_reward_func/std": 23.642847061157227, "step": 447 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 430.0, "completions/mean_length": 314.0625, "completions/mean_terminated_length": 210.38095092773438, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "epoch": 0.3584, "grad_norm": 16.456398010253906, "kl": 2.2950439453125, "learning_rate": 1e-06, "loss": 0.3137, "num_tokens": 5777040.0, "reward": -8.64706802368164, "reward_std": 6.216998100280762, "rewards/rm_reward_func/mean": -8.64706802368164, "rewards/rm_reward_func/std": 15.454483032226562, "step": 448 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40625, "completions/max_length": 512.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 356.21875, "completions/mean_terminated_length": 249.63157653808594, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "epoch": 0.3592, "grad_norm": 9.984318733215332, "kl": 0.9708251953125, "learning_rate": 1e-06, "loss": 0.1167, "num_tokens": 5791887.0, "reward": 2.573716163635254, "reward_std": 8.430018424987793, "rewards/rm_reward_func/mean": 2.573716163635254, "rewards/rm_reward_func/std": 14.201119422912598, "step": 449 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 406.09375, "completions/mean_terminated_length": 370.79168701171875, "completions/min_length": 110.0, "completions/min_terminated_length": 110.0, "epoch": 0.36, "grad_norm": 7.163238525390625, "kl": 0.5322265625, "learning_rate": 1e-06, "loss": 0.0106, "num_tokens": 5807138.0, "reward": -2.3760223388671875, "reward_std": 7.770718574523926, "rewards/rm_reward_func/mean": -2.3760223388671875, "rewards/rm_reward_func/std": 10.865222930908203, "step": 450 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.46875, "completions/max_length": 512.0, "completions/max_terminated_length": 373.0, "completions/mean_length": 329.5, "completions/mean_terminated_length": 168.47059631347656, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "epoch": 0.3608, "grad_norm": 17.861509323120117, "kl": 5.4453125, "learning_rate": 1e-06, "loss": 0.5361, "num_tokens": 5822826.0, "reward": -9.7861328125, "reward_std": 11.642084121704102, "rewards/rm_reward_func/mean": -9.7861328125, "rewards/rm_reward_func/std": 16.064666748046875, "step": 451 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.625, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 445.1875, "completions/mean_terminated_length": 333.8333435058594, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "epoch": 0.3616, "grad_norm": 37.739498138427734, "kl": 4.68994140625, "learning_rate": 1e-06, "loss": 0.2919, "num_tokens": 5839992.0, "reward": -1.2283172607421875, "reward_std": 15.66540813446045, "rewards/rm_reward_func/mean": -1.2283172607421875, "rewards/rm_reward_func/std": 21.28140640258789, "step": 452 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.59375, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 413.0, "completions/mean_terminated_length": 268.3077087402344, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "epoch": 0.3624, "grad_norm": 15.843510627746582, "kl": 5.333740234375, "learning_rate": 1e-06, "loss": 0.2438, "num_tokens": 5857648.0, "reward": -10.1357421875, "reward_std": 7.069613456726074, "rewards/rm_reward_func/mean": -10.1357421875, "rewards/rm_reward_func/std": 11.879778861999512, "step": 453 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.84375, "completions/max_length": 512.0, "completions/max_terminated_length": 430.0, "completions/mean_length": 492.78125, "completions/mean_terminated_length": 389.0, "completions/min_length": 257.0, "completions/min_terminated_length": 257.0, "epoch": 0.3632, "grad_norm": 7.660159587860107, "kl": 6.46484375, "learning_rate": 1e-06, "loss": 0.3101, "num_tokens": 5875665.0, "reward": -10.20697021484375, "reward_std": 12.935501098632812, "rewards/rm_reward_func/mean": -10.20697021484375, "rewards/rm_reward_func/std": 14.417232513427734, "step": 454 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.6875, "completions/max_length": 512.0, "completions/max_terminated_length": 267.0, "completions/mean_length": 411.5, "completions/mean_terminated_length": 190.40000915527344, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "epoch": 0.364, "grad_norm": 12.648509979248047, "kl": 8.0, "learning_rate": 1e-06, "loss": 0.625, "num_tokens": 5891249.0, "reward": -18.5673828125, "reward_std": 8.711212158203125, "rewards/rm_reward_func/mean": -18.5673828125, "rewards/rm_reward_func/std": 9.04566478729248, "step": 455 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 512.0, "completions/max_terminated_length": 403.0, "completions/mean_length": 350.40625, "completions/mean_terminated_length": 188.8125, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "epoch": 0.3648, "grad_norm": 6.931374549865723, "kl": 7.671875, "learning_rate": 1e-06, "loss": 0.6147, "num_tokens": 5906094.0, "reward": -11.6011962890625, "reward_std": 11.895530700683594, "rewards/rm_reward_func/mean": -11.6011962890625, "rewards/rm_reward_func/std": 14.706385612487793, "step": 456 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.53125, "completions/max_length": 512.0, "completions/max_terminated_length": 473.0, "completions/mean_length": 411.875, "completions/mean_terminated_length": 298.4000244140625, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "epoch": 0.3656, "grad_norm": 7.6649980545043945, "kl": 6.799072265625, "learning_rate": 1e-06, "loss": 0.4754, "num_tokens": 5921498.0, "reward": -13.50347900390625, "reward_std": 9.386650085449219, "rewards/rm_reward_func/mean": -13.50347900390625, "rewards/rm_reward_func/std": 14.427614212036133, "step": 457 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.4375, "completions/max_length": 512.0, "completions/max_terminated_length": 475.0, "completions/mean_length": 412.59375, "completions/mean_terminated_length": 335.27777099609375, "completions/min_length": 184.0, "completions/min_terminated_length": 184.0, "epoch": 0.3664, "grad_norm": 14.270245552062988, "kl": 3.159912109375, "learning_rate": 1e-06, "loss": 0.2522, "num_tokens": 5939749.0, "reward": 0.822509765625, "reward_std": 10.554486274719238, "rewards/rm_reward_func/mean": 0.822509765625, "rewards/rm_reward_func/std": 19.273405075073242, "step": 458 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 478.0, "completions/mean_length": 436.84375, "completions/mean_terminated_length": 391.75, "completions/min_length": 145.0, "completions/min_terminated_length": 145.0, "epoch": 0.3672, "grad_norm": 7.888670921325684, "kl": 3.0789794921875, "learning_rate": 1e-06, "loss": 0.1706, "num_tokens": 5957832.0, "reward": -1.209869384765625, "reward_std": 4.854766368865967, "rewards/rm_reward_func/mean": -1.209869384765625, "rewards/rm_reward_func/std": 25.843124389648438, "step": 459 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.625, "completions/max_length": 512.0, "completions/max_terminated_length": 477.0, "completions/mean_length": 439.1875, "completions/mean_terminated_length": 317.8333435058594, "completions/min_length": 90.0, "completions/min_terminated_length": 90.0, "epoch": 0.368, "grad_norm": 17.467594146728516, "kl": 6.8984375, "learning_rate": 1e-06, "loss": 0.4475, "num_tokens": 5975382.0, "reward": -8.3388671875, "reward_std": 10.500482559204102, "rewards/rm_reward_func/mean": -8.3388671875, "rewards/rm_reward_func/std": 15.440128326416016, "step": 460 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 373.3125, "completions/mean_terminated_length": 334.47998046875, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "epoch": 0.3688, "grad_norm": 19.844507217407227, "kl": 2.7052001953125, "learning_rate": 1e-06, "loss": 0.379, "num_tokens": 5990128.0, "reward": 8.0970458984375, "reward_std": 15.225787162780762, "rewards/rm_reward_func/mean": 8.0970458984375, "rewards/rm_reward_func/std": 18.16506004333496, "step": 461 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 446.375, "completions/mean_terminated_length": 412.0, "completions/min_length": 232.0, "completions/min_terminated_length": 232.0, "epoch": 0.3696, "grad_norm": 7.17181921005249, "kl": 0.4322509765625, "learning_rate": 1e-06, "loss": 0.0593, "num_tokens": 6007132.0, "reward": 8.548095703125, "reward_std": 7.690432548522949, "rewards/rm_reward_func/mean": 8.548095703125, "rewards/rm_reward_func/std": 8.413957595825195, "step": 462 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.53125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 439.125, "completions/mean_terminated_length": 356.5333557128906, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "epoch": 0.3704, "grad_norm": 5.347095012664795, "kl": 6.09375, "learning_rate": 1e-06, "loss": 0.3736, "num_tokens": 6025032.0, "reward": 1.619140625, "reward_std": 15.041004180908203, "rewards/rm_reward_func/mean": 1.619140625, "rewards/rm_reward_func/std": 28.54711151123047, "step": 463 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 247.9375, "completions/mean_terminated_length": 239.41934204101562, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "epoch": 0.3712, "grad_norm": 10.885205268859863, "kl": 1.0289306640625, "learning_rate": 1e-06, "loss": 0.0564, "num_tokens": 6035230.0, "reward": 15.275390625, "reward_std": 8.497432708740234, "rewards/rm_reward_func/mean": 15.275390625, "rewards/rm_reward_func/std": 14.570096015930176, "step": 464 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 474.0, "completions/mean_length": 345.34375, "completions/mean_terminated_length": 314.4814758300781, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "epoch": 0.372, "grad_norm": 16.305341720581055, "kl": 2.215576171875, "learning_rate": 1e-06, "loss": 0.2026, "num_tokens": 6048537.0, "reward": 1.36273193359375, "reward_std": 9.692459106445312, "rewards/rm_reward_func/mean": 1.36273193359375, "rewards/rm_reward_func/std": 17.84818458557129, "step": 465 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 407.4375, "completions/mean_terminated_length": 302.875, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "epoch": 0.3728, "grad_norm": 7.994433879852295, "kl": 3.80224609375, "learning_rate": 1e-06, "loss": 0.2663, "num_tokens": 6064695.0, "reward": 3.369873046875, "reward_std": 8.856344223022461, "rewards/rm_reward_func/mean": 3.369873046875, "rewards/rm_reward_func/std": 22.26304817199707, "step": 466 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 458.0, "completions/mean_length": 290.5, "completions/mean_terminated_length": 283.3548278808594, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.3736, "grad_norm": 6.062528133392334, "kl": 0.313720703125, "learning_rate": 1e-06, "loss": 0.0317, "num_tokens": 6077287.0, "reward": 5.29248046875, "reward_std": 8.455047607421875, "rewards/rm_reward_func/mean": 5.29248046875, "rewards/rm_reward_func/std": 13.72189998626709, "step": 467 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 396.28125, "completions/mean_terminated_length": 326.8500061035156, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "epoch": 0.3744, "grad_norm": 13.018776893615723, "kl": 3.998291015625, "learning_rate": 1e-06, "loss": 0.3904, "num_tokens": 6092112.0, "reward": 0.782562255859375, "reward_std": 9.454818725585938, "rewards/rm_reward_func/mean": 0.782562255859375, "rewards/rm_reward_func/std": 14.342537879943848, "step": 468 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 457.0, "completions/mean_terminated_length": 428.19049072265625, "completions/min_length": 363.0, "completions/min_terminated_length": 363.0, "epoch": 0.3752, "grad_norm": 7.1996026039123535, "kl": 3.93896484375, "learning_rate": 1e-06, "loss": 0.1778, "num_tokens": 6109216.0, "reward": 5.27410888671875, "reward_std": 14.364347457885742, "rewards/rm_reward_func/mean": 5.27410888671875, "rewards/rm_reward_func/std": 22.558135986328125, "step": 469 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.4375, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 405.4375, "completions/mean_terminated_length": 322.5555725097656, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "epoch": 0.376, "grad_norm": 19.028276443481445, "kl": 8.43359375, "learning_rate": 1e-06, "loss": 0.5417, "num_tokens": 6124742.0, "reward": -0.470703125, "reward_std": 12.874818801879883, "rewards/rm_reward_func/mean": -0.470703125, "rewards/rm_reward_func/std": 20.95646095275879, "step": 470 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40625, "completions/max_length": 512.0, "completions/max_terminated_length": 398.0, "completions/mean_length": 392.8125, "completions/mean_terminated_length": 311.2631530761719, "completions/min_length": 217.0, "completions/min_terminated_length": 217.0, "epoch": 0.3768, "grad_norm": 28.145797729492188, "kl": 6.830322265625, "learning_rate": 1e-06, "loss": 0.4005, "num_tokens": 6141808.0, "reward": -3.615478515625, "reward_std": 9.015436172485352, "rewards/rm_reward_func/mean": -3.615478515625, "rewards/rm_reward_func/std": 18.243576049804688, "step": 471 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 475.0, "completions/mean_length": 417.125, "completions/mean_terminated_length": 374.0, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "epoch": 0.3776, "grad_norm": 41.16624450683594, "kl": 7.7880859375, "learning_rate": 1e-06, "loss": 0.3244, "num_tokens": 6157436.0, "reward": 10.93359375, "reward_std": 15.283958435058594, "rewards/rm_reward_func/mean": 10.93359375, "rewards/rm_reward_func/std": 27.59185028076172, "step": 472 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 492.0, "completions/mean_length": 351.3125, "completions/mean_terminated_length": 278.2727355957031, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "epoch": 0.3784, "grad_norm": 19.67943572998047, "kl": 7.505615234375, "learning_rate": 1e-06, "loss": 0.536, "num_tokens": 6173246.0, "reward": 1.380859375, "reward_std": 14.768939971923828, "rewards/rm_reward_func/mean": 1.380859375, "rewards/rm_reward_func/std": 18.74730682373047, "step": 473 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40625, "completions/max_length": 512.0, "completions/max_terminated_length": 482.0, "completions/mean_length": 406.28125, "completions/mean_terminated_length": 333.9473571777344, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "epoch": 0.3792, "grad_norm": 14.080571174621582, "kl": 6.677734375, "learning_rate": 1e-06, "loss": 0.3526, "num_tokens": 6189295.0, "reward": -8.592529296875, "reward_std": 8.344656944274902, "rewards/rm_reward_func/mean": -8.592529296875, "rewards/rm_reward_func/std": 14.625808715820312, "step": 474 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 372.5625, "completions/mean_terminated_length": 326.0833435058594, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "epoch": 0.38, "grad_norm": 55.990806579589844, "kl": 3.190673828125, "learning_rate": 1e-06, "loss": 0.3699, "num_tokens": 6204681.0, "reward": 10.92529296875, "reward_std": 17.398414611816406, "rewards/rm_reward_func/mean": 10.92529296875, "rewards/rm_reward_func/std": 21.6011905670166, "step": 475 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 473.0, "completions/mean_length": 336.34375, "completions/mean_terminated_length": 303.8148193359375, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "epoch": 0.3808, "grad_norm": 26.353031158447266, "kl": 3.114013671875, "learning_rate": 1e-06, "loss": 0.2881, "num_tokens": 6217324.0, "reward": -5.37896728515625, "reward_std": 7.184835433959961, "rewards/rm_reward_func/mean": -5.37896728515625, "rewards/rm_reward_func/std": 18.45599365234375, "step": 476 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 417.75, "completions/mean_terminated_length": 396.0, "completions/min_length": 193.0, "completions/min_terminated_length": 193.0, "epoch": 0.3816, "grad_norm": 8.078795433044434, "kl": 3.979248046875, "learning_rate": 1e-06, "loss": 0.2362, "num_tokens": 6233940.0, "reward": 8.0015869140625, "reward_std": 12.765542984008789, "rewards/rm_reward_func/mean": 8.0015869140625, "rewards/rm_reward_func/std": 16.40848159790039, "step": 477 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 393.375, "completions/mean_terminated_length": 322.20001220703125, "completions/min_length": 188.0, "completions/min_terminated_length": 188.0, "epoch": 0.3824, "grad_norm": 16.03205108642578, "kl": 7.90283203125, "learning_rate": 1e-06, "loss": 0.3989, "num_tokens": 6249568.0, "reward": -4.557373046875, "reward_std": 9.42959213256836, "rewards/rm_reward_func/mean": -4.557373046875, "rewards/rm_reward_func/std": 19.618621826171875, "step": 478 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 414.0, "completions/mean_length": 391.4375, "completions/mean_terminated_length": 328.28570556640625, "completions/min_length": 165.0, "completions/min_terminated_length": 165.0, "epoch": 0.3832, "grad_norm": 21.212261199951172, "kl": 7.78076171875, "learning_rate": 1e-06, "loss": 0.364, "num_tokens": 6264534.0, "reward": 0.4144287109375, "reward_std": 8.546712875366211, "rewards/rm_reward_func/mean": 0.4144287109375, "rewards/rm_reward_func/std": 16.865026473999023, "step": 479 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 370.65625, "completions/mean_terminated_length": 323.54168701171875, "completions/min_length": 219.0, "completions/min_terminated_length": 219.0, "epoch": 0.384, "grad_norm": 25.12664031982422, "kl": 6.88818359375, "learning_rate": 1e-06, "loss": 0.3655, "num_tokens": 6278235.0, "reward": 4.96453857421875, "reward_std": 17.571796417236328, "rewards/rm_reward_func/mean": 4.96453857421875, "rewards/rm_reward_func/std": 19.67218589782715, "step": 480 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.6875, "completions/max_length": 512.0, "completions/max_terminated_length": 209.0, "completions/mean_length": 390.875, "completions/mean_terminated_length": 124.4000015258789, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "epoch": 0.3848, "grad_norm": 79.35224914550781, "kl": 21.0, "learning_rate": 1e-06, "loss": 1.0686, "num_tokens": 6294447.0, "reward": -10.9208984375, "reward_std": 13.471847534179688, "rewards/rm_reward_func/mean": -10.9208984375, "rewards/rm_reward_func/std": 27.3441219329834, "step": 481 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.6875, "completions/max_length": 512.0, "completions/max_terminated_length": 229.0, "completions/mean_length": 404.0, "completions/mean_terminated_length": 166.40000915527344, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "epoch": 0.3856, "grad_norm": 132.53387451171875, "kl": 13.1297607421875, "learning_rate": 1e-06, "loss": 0.7436, "num_tokens": 6311167.0, "reward": -12.4658203125, "reward_std": 8.332673072814941, "rewards/rm_reward_func/mean": -12.4658203125, "rewards/rm_reward_func/std": 14.378026962280273, "step": 482 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.46875, "completions/max_length": 512.0, "completions/max_terminated_length": 486.0, "completions/mean_length": 379.8125, "completions/mean_terminated_length": 263.1764831542969, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "epoch": 0.3864, "grad_norm": 17.89412498474121, "kl": 7.271484375, "learning_rate": 1e-06, "loss": 0.4685, "num_tokens": 6325729.0, "reward": -2.525604248046875, "reward_std": 10.174657821655273, "rewards/rm_reward_func/mean": -2.525604248046875, "rewards/rm_reward_func/std": 19.22846221923828, "step": 483 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 355.34375, "completions/mean_terminated_length": 326.3333435058594, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "epoch": 0.3872, "grad_norm": 10.339418411254883, "kl": 5.09521484375, "learning_rate": 1e-06, "loss": 0.349, "num_tokens": 6339132.0, "reward": 0.1297607421875, "reward_std": 12.494827270507812, "rewards/rm_reward_func/mean": 0.1297607421875, "rewards/rm_reward_func/std": 17.290435791015625, "step": 484 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.59375, "completions/max_length": 512.0, "completions/max_terminated_length": 388.0, "completions/mean_length": 372.75, "completions/mean_terminated_length": 169.23077392578125, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "epoch": 0.388, "grad_norm": 12.62641429901123, "kl": 8.357177734375, "learning_rate": 1e-06, "loss": 0.6359, "num_tokens": 6357132.0, "reward": -7.014617919921875, "reward_std": 11.388704299926758, "rewards/rm_reward_func/mean": -7.014617919921875, "rewards/rm_reward_func/std": 11.901640892028809, "step": 485 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 450.0, "completions/mean_length": 307.125, "completions/mean_terminated_length": 214.0, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "epoch": 0.3888, "grad_norm": 13.34914779663086, "kl": 6.413330078125, "learning_rate": 1e-06, "loss": 0.5975, "num_tokens": 6369224.0, "reward": -6.581207275390625, "reward_std": 14.152790069580078, "rewards/rm_reward_func/mean": -6.581207275390625, "rewards/rm_reward_func/std": 17.590269088745117, "step": 486 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 476.0, "completions/mean_length": 422.0625, "completions/mean_terminated_length": 368.1000061035156, "completions/min_length": 241.0, "completions/min_terminated_length": 241.0, "epoch": 0.3896, "grad_norm": 15.647866249084473, "kl": 5.72802734375, "learning_rate": 1e-06, "loss": 0.3176, "num_tokens": 6385186.0, "reward": 2.65093994140625, "reward_std": 8.202743530273438, "rewards/rm_reward_func/mean": 2.65093994140625, "rewards/rm_reward_func/std": 15.850854873657227, "step": 487 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 359.40625, "completions/mean_terminated_length": 279.4761962890625, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "epoch": 0.3904, "grad_norm": 43.07307434082031, "kl": 6.449462890625, "learning_rate": 1e-06, "loss": 0.5931, "num_tokens": 6400463.0, "reward": -1.673583984375, "reward_std": 10.083917617797852, "rewards/rm_reward_func/mean": -1.673583984375, "rewards/rm_reward_func/std": 13.745939254760742, "step": 488 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 512.0, "completions/max_terminated_length": 394.0, "completions/mean_length": 399.375, "completions/mean_terminated_length": 286.75, "completions/min_length": 110.0, "completions/min_terminated_length": 110.0, "epoch": 0.3912, "grad_norm": 91.3353042602539, "kl": 7.009765625, "learning_rate": 1e-06, "loss": 0.4609, "num_tokens": 6415571.0, "reward": -7.83154296875, "reward_std": 9.295122146606445, "rewards/rm_reward_func/mean": -7.83154296875, "rewards/rm_reward_func/std": 16.444271087646484, "step": 489 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 484.0, "completions/mean_length": 362.71875, "completions/mean_terminated_length": 328.2692565917969, "completions/min_length": 133.0, "completions/min_terminated_length": 133.0, "epoch": 0.392, "grad_norm": 24.025558471679688, "kl": 3.818359375, "learning_rate": 1e-06, "loss": 0.3076, "num_tokens": 6429994.0, "reward": 12.959228515625, "reward_std": 14.361933708190918, "rewards/rm_reward_func/mean": 12.959228515625, "rewards/rm_reward_func/std": 26.099708557128906, "step": 490 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 512.0, "completions/max_terminated_length": 419.0, "completions/mean_length": 348.15625, "completions/mean_terminated_length": 184.3125, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "epoch": 0.3928, "grad_norm": 49.87342071533203, "kl": 9.898193359375, "learning_rate": 1e-06, "loss": 0.8155, "num_tokens": 6449151.0, "reward": -8.174560546875, "reward_std": 11.404885292053223, "rewards/rm_reward_func/mean": -8.174560546875, "rewards/rm_reward_func/std": 15.167132377624512, "step": 491 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 308.5, "completions/mean_terminated_length": 270.8148193359375, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "epoch": 0.3936, "grad_norm": 42.138450622558594, "kl": 4.73828125, "learning_rate": 1e-06, "loss": 0.5528, "num_tokens": 6463335.0, "reward": 11.81298828125, "reward_std": 15.574457168579102, "rewards/rm_reward_func/mean": 11.81298828125, "rewards/rm_reward_func/std": 20.5043888092041, "step": 492 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 486.0, "completions/mean_length": 285.34375, "completions/mean_terminated_length": 166.61904907226562, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "epoch": 0.3944, "grad_norm": 122.83219146728516, "kl": 9.7734375, "learning_rate": 1e-06, "loss": 0.9872, "num_tokens": 6475618.0, "reward": -2.90985107421875, "reward_std": 13.12482738494873, "rewards/rm_reward_func/mean": -2.90985107421875, "rewards/rm_reward_func/std": 14.76265811920166, "step": 493 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 352.0625, "completions/mean_terminated_length": 256.1000061035156, "completions/min_length": 82.0, "completions/min_terminated_length": 82.0, "epoch": 0.3952, "grad_norm": 18.307031631469727, "kl": 11.32470703125, "learning_rate": 1e-06, "loss": 0.8083, "num_tokens": 6489324.0, "reward": -12.8271484375, "reward_std": 8.004693984985352, "rewards/rm_reward_func/mean": -12.8271484375, "rewards/rm_reward_func/std": 11.452200889587402, "step": 494 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 361.25, "completions/mean_terminated_length": 302.2608642578125, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "epoch": 0.396, "grad_norm": 24.6053409576416, "kl": 6.4337158203125, "learning_rate": 1e-06, "loss": 0.2939, "num_tokens": 6507252.0, "reward": -5.3721923828125, "reward_std": 7.6330060958862305, "rewards/rm_reward_func/mean": -5.3721923828125, "rewards/rm_reward_func/std": 12.24431324005127, "step": 495 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 278.3125, "completions/mean_terminated_length": 200.4166717529297, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "epoch": 0.3968, "grad_norm": 15.430408477783203, "kl": 4.6640625, "learning_rate": 1e-06, "loss": 0.4849, "num_tokens": 6521006.0, "reward": 11.46609115600586, "reward_std": 17.760250091552734, "rewards/rm_reward_func/mean": 11.46609115600586, "rewards/rm_reward_func/std": 25.725770950317383, "step": 496 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.6875, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 470.03125, "completions/mean_terminated_length": 377.70001220703125, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "epoch": 0.3976, "grad_norm": 171.98805236816406, "kl": 15.0703125, "learning_rate": 1e-06, "loss": 0.7248, "num_tokens": 6539287.0, "reward": -14.515625, "reward_std": 12.340795516967773, "rewards/rm_reward_func/mean": -14.515625, "rewards/rm_reward_func/std": 17.410518646240234, "step": 497 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.6875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 479.625, "completions/mean_terminated_length": 408.3999938964844, "completions/min_length": 275.0, "completions/min_terminated_length": 275.0, "epoch": 0.3984, "grad_norm": 38.708248138427734, "kl": 13.7109375, "learning_rate": 1e-06, "loss": 0.6117, "num_tokens": 6556987.0, "reward": -11.59375, "reward_std": 13.794965744018555, "rewards/rm_reward_func/mean": -11.59375, "rewards/rm_reward_func/std": 23.78890609741211, "step": 498 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.625, "completions/max_length": 512.0, "completions/max_terminated_length": 421.0, "completions/mean_length": 409.3125, "completions/mean_terminated_length": 238.1666717529297, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "epoch": 0.3992, "grad_norm": 25.57811164855957, "kl": 14.84375, "learning_rate": 1e-06, "loss": 0.8756, "num_tokens": 6572293.0, "reward": -6.9130859375, "reward_std": 14.296964645385742, "rewards/rm_reward_func/mean": -6.9130859375, "rewards/rm_reward_func/std": 21.80596351623535, "step": 499 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.4375, "completions/max_length": 512.0, "completions/max_terminated_length": 485.0, "completions/mean_length": 378.0, "completions/mean_terminated_length": 273.77777099609375, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "epoch": 0.4, "grad_norm": 70.98407745361328, "kl": 7.545654296875, "learning_rate": 1e-06, "loss": 0.5359, "num_tokens": 6586981.0, "reward": -2.8328094482421875, "reward_std": 7.248239040374756, "rewards/rm_reward_func/mean": -2.8328094482421875, "rewards/rm_reward_func/std": 12.194914817810059, "step": 500 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.8125, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 496.78125, "completions/mean_terminated_length": 430.8333435058594, "completions/min_length": 373.0, "completions/min_terminated_length": 373.0, "epoch": 0.4008, "grad_norm": 9.479290008544922, "kl": 6.978515625, "learning_rate": 1e-06, "loss": 0.3081, "num_tokens": 6606366.0, "reward": -7.8212890625, "reward_std": 11.259090423583984, "rewards/rm_reward_func/mean": -7.8212890625, "rewards/rm_reward_func/std": 18.525226593017578, "step": 501 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.46875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 385.9375, "completions/mean_terminated_length": 274.70587158203125, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "epoch": 0.4016, "grad_norm": 42.01311492919922, "kl": 5.5234375, "learning_rate": 1e-06, "loss": 0.5541, "num_tokens": 6621700.0, "reward": -0.1533203125, "reward_std": 14.605756759643555, "rewards/rm_reward_func/mean": -0.1533203125, "rewards/rm_reward_func/std": 23.270185470581055, "step": 502 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.71875, "completions/max_length": 512.0, "completions/max_terminated_length": 454.0, "completions/mean_length": 436.625, "completions/mean_terminated_length": 244.0, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "epoch": 0.4024, "grad_norm": 20.074079513549805, "kl": 7.421875, "learning_rate": 1e-06, "loss": 0.5378, "num_tokens": 6637896.0, "reward": -12.885040283203125, "reward_std": 12.589856147766113, "rewards/rm_reward_func/mean": -12.885040283203125, "rewards/rm_reward_func/std": 16.110185623168945, "step": 503 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 231.0, "completions/mean_length": 243.875, "completions/mean_terminated_length": 103.42857360839844, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "epoch": 0.4032, "grad_norm": 24.548263549804688, "kl": 8.2744140625, "learning_rate": 1e-06, "loss": 0.9126, "num_tokens": 6648780.0, "reward": -6.2499847412109375, "reward_std": 12.00403118133545, "rewards/rm_reward_func/mean": -6.2499847412109375, "rewards/rm_reward_func/std": 13.340556144714355, "step": 504 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.59375, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 460.0625, "completions/mean_terminated_length": 384.15386962890625, "completions/min_length": 214.0, "completions/min_terminated_length": 214.0, "epoch": 0.404, "grad_norm": 17.74894905090332, "kl": 4.763671875, "learning_rate": 1e-06, "loss": 0.2654, "num_tokens": 6666302.0, "reward": -6.497314453125, "reward_std": 12.197681427001953, "rewards/rm_reward_func/mean": -6.497314453125, "rewards/rm_reward_func/std": 15.69522476196289, "step": 505 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5625, "completions/max_length": 512.0, "completions/max_terminated_length": 426.0, "completions/mean_length": 376.0, "completions/mean_terminated_length": 201.1428680419922, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "epoch": 0.4048, "grad_norm": 13.425418853759766, "kl": 10.6875, "learning_rate": 1e-06, "loss": 0.8323, "num_tokens": 6683206.0, "reward": -5.591552734375, "reward_std": 16.38616943359375, "rewards/rm_reward_func/mean": -5.591552734375, "rewards/rm_reward_func/std": 19.70439910888672, "step": 506 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.6875, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 437.09375, "completions/mean_terminated_length": 272.3000183105469, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "epoch": 0.4056, "grad_norm": 23.703475952148438, "kl": 9.7734375, "learning_rate": 1e-06, "loss": 0.5968, "num_tokens": 6700297.0, "reward": -4.63623046875, "reward_std": 19.530458450317383, "rewards/rm_reward_func/mean": -4.63623046875, "rewards/rm_reward_func/std": 19.052122116088867, "step": 507 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.46875, "completions/max_length": 512.0, "completions/max_terminated_length": 435.0, "completions/mean_length": 324.25, "completions/mean_terminated_length": 158.58824157714844, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "epoch": 0.4064, "grad_norm": 28.222517013549805, "kl": 9.54296875, "learning_rate": 1e-06, "loss": 0.5761, "num_tokens": 6715521.0, "reward": -10.552978515625, "reward_std": 8.230770111083984, "rewards/rm_reward_func/mean": -10.552978515625, "rewards/rm_reward_func/std": 15.682586669921875, "step": 508 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "epoch": 0.4072, "grad_norm": 38.62055969238281, "kl": 15.5, "learning_rate": 1e-06, "loss": 0.6205, "num_tokens": 6733761.0, "reward": -21.97564697265625, "reward_std": 5.978219985961914, "rewards/rm_reward_func/mean": -21.97564697265625, "rewards/rm_reward_func/std": 7.307789325714111, "step": 509 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.4375, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 366.53125, "completions/mean_terminated_length": 253.38888549804688, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.408, "grad_norm": 25.734407424926758, "kl": 8.80078125, "learning_rate": 1e-06, "loss": 0.6397, "num_tokens": 6749586.0, "reward": -2.8720703125, "reward_std": 11.287141799926758, "rewards/rm_reward_func/mean": -2.8720703125, "rewards/rm_reward_func/std": 18.167095184326172, "step": 510 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.59375, "completions/max_length": 512.0, "completions/max_terminated_length": 449.0, "completions/mean_length": 436.15625, "completions/mean_terminated_length": 325.3077087402344, "completions/min_length": 231.0, "completions/min_terminated_length": 231.0, "epoch": 0.4088, "grad_norm": 21.79092025756836, "kl": 8.068603515625, "learning_rate": 1e-06, "loss": 0.3422, "num_tokens": 6765663.0, "reward": -13.787574768066406, "reward_std": 6.30387020111084, "rewards/rm_reward_func/mean": -13.787574768066406, "rewards/rm_reward_func/std": 17.804723739624023, "step": 511 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.71875, "completions/max_length": 512.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 420.71875, "completions/mean_terminated_length": 187.44444274902344, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "epoch": 0.4096, "grad_norm": 161.0446319580078, "kl": 9.1796875, "learning_rate": 1e-06, "loss": 0.6176, "num_tokens": 6782334.0, "reward": -11.8310546875, "reward_std": 9.512861251831055, "rewards/rm_reward_func/mean": -11.8310546875, "rewards/rm_reward_func/std": 16.280780792236328, "step": 512 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.59375, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 397.84375, "completions/mean_terminated_length": 231.00001525878906, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "epoch": 0.4104, "grad_norm": 12.210495948791504, "kl": 4.85791015625, "learning_rate": 1e-06, "loss": 0.4964, "num_tokens": 6798361.0, "reward": 0.423828125, "reward_std": 10.785526275634766, "rewards/rm_reward_func/mean": 0.423828125, "rewards/rm_reward_func/std": 16.22878074645996, "step": 513 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 384.9375, "completions/mean_terminated_length": 298.0, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.4112, "grad_norm": 27.765371322631836, "kl": 5.994140625, "learning_rate": 1e-06, "loss": 0.5379, "num_tokens": 6813559.0, "reward": -0.2978515625, "reward_std": 15.280010223388672, "rewards/rm_reward_func/mean": -0.2978515625, "rewards/rm_reward_func/std": 18.968994140625, "step": 514 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.75, "completions/max_length": 512.0, "completions/max_terminated_length": 373.0, "completions/mean_length": 456.125, "completions/mean_terminated_length": 288.5, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "epoch": 0.412, "grad_norm": 17.61673927307129, "kl": 7.875, "learning_rate": 1e-06, "loss": 0.4006, "num_tokens": 6832075.0, "reward": -14.86328125, "reward_std": 6.095500946044922, "rewards/rm_reward_func/mean": -14.86328125, "rewards/rm_reward_func/std": 16.34231948852539, "step": 515 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.8125, "completions/max_length": 512.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 480.84375, "completions/mean_terminated_length": 345.8333435058594, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "epoch": 0.4128, "grad_norm": 17.899005889892578, "kl": 6.27734375, "learning_rate": 1e-06, "loss": 0.2511, "num_tokens": 6849974.0, "reward": -14.375091552734375, "reward_std": 6.907338619232178, "rewards/rm_reward_func/mean": -14.375091552734375, "rewards/rm_reward_func/std": 9.020212173461914, "step": 516 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5625, "completions/max_length": 512.0, "completions/max_terminated_length": 268.0, "completions/mean_length": 378.34375, "completions/mean_terminated_length": 206.50001525878906, "completions/min_length": 158.0, "completions/min_terminated_length": 158.0, "epoch": 0.4136, "grad_norm": 25.587743759155273, "kl": 5.07421875, "learning_rate": 1e-06, "loss": 0.3087, "num_tokens": 6866065.0, "reward": -9.5302734375, "reward_std": 5.052003860473633, "rewards/rm_reward_func/mean": -9.5302734375, "rewards/rm_reward_func/std": 10.77525520324707, "step": 517 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.59375, "completions/max_length": 512.0, "completions/max_terminated_length": 491.0, "completions/mean_length": 383.625, "completions/mean_terminated_length": 196.0, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "epoch": 0.4144, "grad_norm": 28.90671730041504, "kl": 4.85546875, "learning_rate": 1e-06, "loss": 0.5353, "num_tokens": 6883749.0, "reward": -7.447265625, "reward_std": 11.519471168518066, "rewards/rm_reward_func/mean": -7.447265625, "rewards/rm_reward_func/std": 13.736445426940918, "step": 518 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.71875, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 460.6875, "completions/mean_terminated_length": 329.5555725097656, "completions/min_length": 205.0, "completions/min_terminated_length": 205.0, "epoch": 0.4152, "grad_norm": 6.363600254058838, "kl": 5.875, "learning_rate": 1e-06, "loss": 0.3459, "num_tokens": 6901867.0, "reward": -8.869140625, "reward_std": 12.279427528381348, "rewards/rm_reward_func/mean": -8.869140625, "rewards/rm_reward_func/std": 18.361650466918945, "step": 519 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5625, "completions/max_length": 512.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 426.90625, "completions/mean_terminated_length": 317.5, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "epoch": 0.416, "grad_norm": 12.628585815429688, "kl": 3.195068359375, "learning_rate": 1e-06, "loss": 0.2452, "num_tokens": 6918720.0, "reward": -3.0980224609375, "reward_std": 8.710382461547852, "rewards/rm_reward_func/mean": -3.0980224609375, "rewards/rm_reward_func/std": 11.913897514343262, "step": 520 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 423.0, "completions/mean_length": 296.28125, "completions/mean_terminated_length": 166.85000610351562, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "epoch": 0.4168, "grad_norm": 17.954702377319336, "kl": 5.048828125, "learning_rate": 1e-06, "loss": 0.4875, "num_tokens": 6931665.0, "reward": 0.85546875, "reward_std": 10.385427474975586, "rewards/rm_reward_func/mean": 0.85546875, "rewards/rm_reward_func/std": 15.259444236755371, "step": 521 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5625, "completions/max_length": 512.0, "completions/max_terminated_length": 447.0, "completions/mean_length": 409.40625, "completions/mean_terminated_length": 277.5, "completions/min_length": 115.0, "completions/min_terminated_length": 115.0, "epoch": 0.4176, "grad_norm": 19.831439971923828, "kl": 6.078125, "learning_rate": 1e-06, "loss": 0.4218, "num_tokens": 6946854.0, "reward": -6.93914794921875, "reward_std": 13.028332710266113, "rewards/rm_reward_func/mean": -6.93914794921875, "rewards/rm_reward_func/std": 14.581646919250488, "step": 522 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 512.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 438.09375, "completions/mean_terminated_length": 364.1875, "completions/min_length": 181.0, "completions/min_terminated_length": 181.0, "epoch": 0.4184, "grad_norm": 30.67502212524414, "kl": 4.91064453125, "learning_rate": 1e-06, "loss": 0.2753, "num_tokens": 6963185.0, "reward": -6.296905517578125, "reward_std": 5.502427101135254, "rewards/rm_reward_func/mean": -6.296905517578125, "rewards/rm_reward_func/std": 12.138696670532227, "step": 523 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5625, "completions/max_length": 512.0, "completions/max_terminated_length": 322.0, "completions/mean_length": 346.78125, "completions/mean_terminated_length": 134.35714721679688, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "epoch": 0.4192, "grad_norm": 33.20919418334961, "kl": 6.63623046875, "learning_rate": 1e-06, "loss": 0.4747, "num_tokens": 6976906.0, "reward": -11.229736328125, "reward_std": 4.540736198425293, "rewards/rm_reward_func/mean": -11.229736328125, "rewards/rm_reward_func/std": 11.639115333557129, "step": 524 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 370.03125, "completions/mean_terminated_length": 343.7407531738281, "completions/min_length": 144.0, "completions/min_terminated_length": 144.0, "epoch": 0.42, "grad_norm": 9.042720794677734, "kl": 0.87451171875, "learning_rate": 1e-06, "loss": 0.174, "num_tokens": 6990763.0, "reward": 4.61993408203125, "reward_std": 7.443538665771484, "rewards/rm_reward_func/mean": 4.61993408203125, "rewards/rm_reward_func/std": 12.067349433898926, "step": 525 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.6875, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 399.65625, "completions/mean_terminated_length": 152.5, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "epoch": 0.4208, "grad_norm": 23.3360538482666, "kl": 12.09375, "learning_rate": 1e-06, "loss": 0.8084, "num_tokens": 7008024.0, "reward": -11.912353515625, "reward_std": 9.951939582824707, "rewards/rm_reward_func/mean": -11.912353515625, "rewards/rm_reward_func/std": 15.001555442810059, "step": 526 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 479.0, "completions/mean_length": 364.9375, "completions/mean_terminated_length": 287.9047546386719, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "epoch": 0.4216, "grad_norm": 11.807538986206055, "kl": 3.63671875, "learning_rate": 1e-06, "loss": 0.2078, "num_tokens": 7023238.0, "reward": 1.132568359375, "reward_std": 11.639840126037598, "rewards/rm_reward_func/mean": 1.132568359375, "rewards/rm_reward_func/std": 15.697065353393555, "step": 527 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.4375, "completions/max_length": 512.0, "completions/max_terminated_length": 325.0, "completions/mean_length": 308.46875, "completions/mean_terminated_length": 150.1666717529297, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "epoch": 0.4224, "grad_norm": 10.746874809265137, "kl": 7.41259765625, "learning_rate": 1e-06, "loss": 0.4985, "num_tokens": 7035693.0, "reward": -4.9210205078125, "reward_std": 7.971255302429199, "rewards/rm_reward_func/mean": -4.9210205078125, "rewards/rm_reward_func/std": 17.77606773376465, "step": 528 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.59375, "completions/max_length": 512.0, "completions/max_terminated_length": 475.0, "completions/mean_length": 450.3125, "completions/mean_terminated_length": 360.15386962890625, "completions/min_length": 290.0, "completions/min_terminated_length": 290.0, "epoch": 0.4232, "grad_norm": 18.503520965576172, "kl": 9.25390625, "learning_rate": 1e-06, "loss": 0.4482, "num_tokens": 7054207.0, "reward": 1.451171875, "reward_std": 22.573057174682617, "rewards/rm_reward_func/mean": 1.451171875, "rewards/rm_reward_func/std": 27.66277313232422, "step": 529 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.84375, "completions/max_length": 512.0, "completions/max_terminated_length": 328.0, "completions/mean_length": 467.78125, "completions/mean_terminated_length": 229.0, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "epoch": 0.424, "grad_norm": 47.28842544555664, "kl": 14.5625, "learning_rate": 1e-06, "loss": 0.7504, "num_tokens": 7072056.0, "reward": -19.1044921875, "reward_std": 12.893132209777832, "rewards/rm_reward_func/mean": -19.1044921875, "rewards/rm_reward_func/std": 14.44466781616211, "step": 530 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.8125, "completions/max_length": 512.0, "completions/max_terminated_length": 424.0, "completions/mean_length": 474.28125, "completions/mean_terminated_length": 310.8333435058594, "completions/min_length": 229.0, "completions/min_terminated_length": 229.0, "epoch": 0.4248, "grad_norm": 44.54637908935547, "kl": 8.7578125, "learning_rate": 1e-06, "loss": 0.4787, "num_tokens": 7089625.0, "reward": -19.9476318359375, "reward_std": 8.834146499633789, "rewards/rm_reward_func/mean": -19.9476318359375, "rewards/rm_reward_func/std": 10.349533081054688, "step": 531 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5625, "completions/max_length": 512.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 366.28125, "completions/mean_terminated_length": 178.92857360839844, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "epoch": 0.4256, "grad_norm": 522.36865234375, "kl": 6.591796875, "learning_rate": 1e-06, "loss": 0.6736, "num_tokens": 7105522.0, "reward": -1.458984375, "reward_std": 9.65514850616455, "rewards/rm_reward_func/mean": -1.458984375, "rewards/rm_reward_func/std": 23.94460678100586, "step": 532 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.625, "completions/max_length": 512.0, "completions/max_terminated_length": 390.0, "completions/mean_length": 395.84375, "completions/mean_terminated_length": 202.25, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "epoch": 0.4264, "grad_norm": 39.247413635253906, "kl": 7.3671875, "learning_rate": 1e-06, "loss": 0.569, "num_tokens": 7120965.0, "reward": -13.78955078125, "reward_std": 11.25367546081543, "rewards/rm_reward_func/mean": -13.78955078125, "rewards/rm_reward_func/std": 15.153644561767578, "step": 533 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.59375, "completions/max_length": 512.0, "completions/max_terminated_length": 474.0, "completions/mean_length": 418.6875, "completions/mean_terminated_length": 282.3077087402344, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "epoch": 0.4272, "grad_norm": 21.322107315063477, "kl": 4.0390625, "learning_rate": 1e-06, "loss": 0.2022, "num_tokens": 7136419.0, "reward": -12.512939453125, "reward_std": 14.791849136352539, "rewards/rm_reward_func/mean": -12.512939453125, "rewards/rm_reward_func/std": 16.32183074951172, "step": 534 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.6875, "completions/max_length": 512.0, "completions/max_terminated_length": 441.0, "completions/mean_length": 434.34375, "completions/mean_terminated_length": 263.5, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "epoch": 0.428, "grad_norm": 18.7730770111084, "kl": 5.703125, "learning_rate": 1e-06, "loss": 0.4185, "num_tokens": 7152758.0, "reward": -8.214715957641602, "reward_std": 9.016678810119629, "rewards/rm_reward_func/mean": -8.214715957641602, "rewards/rm_reward_func/std": 10.963976860046387, "step": 535 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 449.0, "completions/mean_length": 364.15625, "completions/mean_terminated_length": 275.45001220703125, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "epoch": 0.4288, "grad_norm": 440.05767822265625, "kl": 5.39599609375, "learning_rate": 1e-06, "loss": 0.5068, "num_tokens": 7167107.0, "reward": -0.919921875, "reward_std": 10.649211883544922, "rewards/rm_reward_func/mean": -0.919921875, "rewards/rm_reward_func/std": 23.648954391479492, "step": 536 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.53125, "completions/max_length": 512.0, "completions/max_terminated_length": 404.0, "completions/mean_length": 369.59375, "completions/mean_terminated_length": 208.20001220703125, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "epoch": 0.4296, "grad_norm": 12.493600845336914, "kl": 5.7001953125, "learning_rate": 1e-06, "loss": 0.29, "num_tokens": 7181822.0, "reward": -3.5146484375, "reward_std": 7.881195068359375, "rewards/rm_reward_func/mean": -3.5146484375, "rewards/rm_reward_func/std": 15.348002433776855, "step": 537 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.65625, "completions/max_length": 512.0, "completions/max_terminated_length": 305.0, "completions/mean_length": 395.65625, "completions/mean_terminated_length": 173.5454559326172, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "epoch": 0.4304, "grad_norm": 57.40653991699219, "kl": 9.09375, "learning_rate": 1e-06, "loss": 0.7079, "num_tokens": 7201579.0, "reward": -9.9658203125, "reward_std": 12.702253341674805, "rewards/rm_reward_func/mean": -9.9658203125, "rewards/rm_reward_func/std": 15.439082145690918, "step": 538 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.59375, "completions/max_length": 512.0, "completions/max_terminated_length": 296.0, "completions/mean_length": 341.8125, "completions/mean_terminated_length": 93.0769271850586, "completions/min_length": 40.0, "completions/min_terminated_length": 40.0, "epoch": 0.4312, "grad_norm": 15.874788284301758, "kl": 9.1171875, "learning_rate": 1e-06, "loss": 0.6786, "num_tokens": 7218213.0, "reward": -14.662109375, "reward_std": 6.063241004943848, "rewards/rm_reward_func/mean": -14.662109375, "rewards/rm_reward_func/std": 12.366585731506348, "step": 539 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.84375, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 466.03125, "completions/mean_terminated_length": 217.8000030517578, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "epoch": 0.432, "grad_norm": 16.715585708618164, "kl": 8.3671875, "learning_rate": 1e-06, "loss": 0.3327, "num_tokens": 7235414.0, "reward": -15.6708984375, "reward_std": 11.589529991149902, "rewards/rm_reward_func/mean": -15.6708984375, "rewards/rm_reward_func/std": 12.419350624084473, "step": 540 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.53125, "completions/max_length": 512.0, "completions/max_terminated_length": 481.0, "completions/mean_length": 445.1875, "completions/mean_terminated_length": 369.4666748046875, "completions/min_length": 258.0, "completions/min_terminated_length": 258.0, "epoch": 0.4328, "grad_norm": 5.2816667556762695, "kl": 4.4296875, "learning_rate": 1e-06, "loss": 0.2283, "num_tokens": 7252700.0, "reward": -0.094482421875, "reward_std": 17.288135528564453, "rewards/rm_reward_func/mean": -0.094482421875, "rewards/rm_reward_func/std": 24.340282440185547, "step": 541 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40625, "completions/max_length": 512.0, "completions/max_terminated_length": 476.0, "completions/mean_length": 378.9375, "completions/mean_terminated_length": 287.8947448730469, "completions/min_length": 100.0, "completions/min_terminated_length": 100.0, "epoch": 0.4336, "grad_norm": 20.250680923461914, "kl": 4.130859375, "learning_rate": 1e-06, "loss": 0.4542, "num_tokens": 7267650.0, "reward": -4.664306640625, "reward_std": 12.11115837097168, "rewards/rm_reward_func/mean": -4.664306640625, "rewards/rm_reward_func/std": 14.438556671142578, "step": 542 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.65625, "completions/max_length": 512.0, "completions/max_terminated_length": 300.0, "completions/mean_length": 387.53125, "completions/mean_terminated_length": 149.90908813476562, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "epoch": 0.4344, "grad_norm": 24.06378173828125, "kl": 4.3828125, "learning_rate": 1e-06, "loss": 0.4955, "num_tokens": 7282563.0, "reward": -3.005126953125, "reward_std": 11.770925521850586, "rewards/rm_reward_func/mean": -3.005126953125, "rewards/rm_reward_func/std": 15.770807266235352, "step": 543 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 264.90625, "completions/mean_terminated_length": 168.21739196777344, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "epoch": 0.4352, "grad_norm": 27.67669677734375, "kl": 3.830078125, "learning_rate": 1e-06, "loss": 0.5483, "num_tokens": 7293816.0, "reward": -2.7064208984375, "reward_std": 13.272294998168945, "rewards/rm_reward_func/mean": -2.7064208984375, "rewards/rm_reward_func/std": 16.109525680541992, "step": 544 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 311.9375, "completions/mean_terminated_length": 221.0, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "epoch": 0.436, "grad_norm": 20.555767059326172, "kl": 3.757080078125, "learning_rate": 1e-06, "loss": 0.5049, "num_tokens": 7307534.0, "reward": -0.9964599609375, "reward_std": 8.262689590454102, "rewards/rm_reward_func/mean": -0.9964599609375, "rewards/rm_reward_func/std": 15.29664421081543, "step": 545 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 466.0, "completions/mean_length": 331.0625, "completions/mean_terminated_length": 236.2857208251953, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "epoch": 0.4368, "grad_norm": 19.920595169067383, "kl": 3.9521484375, "learning_rate": 1e-06, "loss": 0.0243, "num_tokens": 7327104.0, "reward": -12.331245422363281, "reward_std": 5.995415687561035, "rewards/rm_reward_func/mean": -12.331245422363281, "rewards/rm_reward_func/std": 11.88155746459961, "step": 546 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5625, "completions/max_length": 512.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 444.3125, "completions/mean_terminated_length": 357.2857360839844, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "epoch": 0.4376, "grad_norm": 21.573915481567383, "kl": 4.890625, "learning_rate": 1e-06, "loss": 0.3193, "num_tokens": 7344522.0, "reward": -5.7215576171875, "reward_std": 17.706233978271484, "rewards/rm_reward_func/mean": -5.7215576171875, "rewards/rm_reward_func/std": 20.520912170410156, "step": 547 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 512.0, "completions/max_terminated_length": 460.0, "completions/mean_length": 403.375, "completions/mean_terminated_length": 294.75, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "epoch": 0.4384, "grad_norm": 18.524627685546875, "kl": 5.6640625, "learning_rate": 1e-06, "loss": 0.368, "num_tokens": 7363686.0, "reward": -8.279296875, "reward_std": 14.83357048034668, "rewards/rm_reward_func/mean": -8.279296875, "rewards/rm_reward_func/std": 16.11411476135254, "step": 548 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 485.0, "completions/mean_length": 251.96875, "completions/mean_terminated_length": 214.82144165039062, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "epoch": 0.4392, "grad_norm": 17.9769344329834, "kl": 3.47265625, "learning_rate": 1e-06, "loss": 0.5011, "num_tokens": 7374869.0, "reward": -9.139419555664062, "reward_std": 6.864280700683594, "rewards/rm_reward_func/mean": -9.139419555664062, "rewards/rm_reward_func/std": 13.617587089538574, "step": 549 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 391.0, "completions/mean_length": 335.46875, "completions/mean_terminated_length": 243.0, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "epoch": 0.44, "grad_norm": 12.295485496520996, "kl": 6.3740234375, "learning_rate": 1e-06, "loss": 0.5446, "num_tokens": 7388164.0, "reward": -0.984375, "reward_std": 13.594259262084961, "rewards/rm_reward_func/mean": -0.984375, "rewards/rm_reward_func/std": 17.934398651123047, "step": 550 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.4375, "completions/max_length": 512.0, "completions/max_terminated_length": 414.0, "completions/mean_length": 326.4375, "completions/mean_terminated_length": 182.11111450195312, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "epoch": 0.4408, "grad_norm": 9.252522468566895, "kl": 9.421875, "learning_rate": 1e-06, "loss": 0.8192, "num_tokens": 7401826.0, "reward": -3.5927734375, "reward_std": 14.570196151733398, "rewards/rm_reward_func/mean": -3.5927734375, "rewards/rm_reward_func/std": 18.126157760620117, "step": 551 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 377.0, "completions/mean_length": 281.34375, "completions/mean_terminated_length": 160.52381896972656, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "epoch": 0.4416, "grad_norm": 8.966374397277832, "kl": 9.5185546875, "learning_rate": 1e-06, "loss": 0.6778, "num_tokens": 7414197.0, "reward": -1.685546875, "reward_std": 9.854778289794922, "rewards/rm_reward_func/mean": -1.685546875, "rewards/rm_reward_func/std": 13.30488395690918, "step": 552 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5625, "completions/max_length": 512.0, "completions/max_terminated_length": 398.0, "completions/mean_length": 359.75, "completions/mean_terminated_length": 164.0, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "epoch": 0.4424, "grad_norm": 21.584428787231445, "kl": 13.09375, "learning_rate": 1e-06, "loss": 0.9478, "num_tokens": 7433813.0, "reward": -11.2989501953125, "reward_std": 9.919830322265625, "rewards/rm_reward_func/mean": -11.2989501953125, "rewards/rm_reward_func/std": 11.97014045715332, "step": 553 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.625, "completions/max_length": 512.0, "completions/max_terminated_length": 466.0, "completions/mean_length": 406.59375, "completions/mean_terminated_length": 230.9166717529297, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "epoch": 0.4432, "grad_norm": 63.04143524169922, "kl": 14.703125, "learning_rate": 1e-06, "loss": 0.7666, "num_tokens": 7449552.0, "reward": -16.6484375, "reward_std": 8.551742553710938, "rewards/rm_reward_func/mean": -16.6484375, "rewards/rm_reward_func/std": 16.584556579589844, "step": 554 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.46875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 380.5625, "completions/mean_terminated_length": 264.5882263183594, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.444, "grad_norm": 592.711669921875, "kl": 17.265625, "learning_rate": 1e-06, "loss": 1.0223, "num_tokens": 7465730.0, "reward": -6.50244140625, "reward_std": 14.185935974121094, "rewards/rm_reward_func/mean": -6.50244140625, "rewards/rm_reward_func/std": 18.073198318481445, "step": 555 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 438.0, "completions/mean_length": 319.03125, "completions/mean_terminated_length": 274.5, "completions/min_length": 103.0, "completions/min_terminated_length": 103.0, "epoch": 0.4448, "grad_norm": 13.58010482788086, "kl": 7.970703125, "learning_rate": 1e-06, "loss": 0.3343, "num_tokens": 7478107.0, "reward": -8.147216796875, "reward_std": 11.794363021850586, "rewards/rm_reward_func/mean": -8.147216796875, "rewards/rm_reward_func/std": 15.557478904724121, "step": 556 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 493.0, "completions/mean_length": 273.84375, "completions/mean_terminated_length": 239.82144165039062, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "epoch": 0.4456, "grad_norm": 25.78789520263672, "kl": 7.234619140625, "learning_rate": 1e-06, "loss": 0.3103, "num_tokens": 7492406.0, "reward": 1.95947265625, "reward_std": 10.006596565246582, "rewards/rm_reward_func/mean": 1.95947265625, "rewards/rm_reward_func/std": 15.28310489654541, "step": 557 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 225.6875, "completions/mean_terminated_length": 172.6666717529297, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "epoch": 0.4464, "grad_norm": 16.13897705078125, "kl": 12.6171875, "learning_rate": 1e-06, "loss": 1.0411, "num_tokens": 7504660.0, "reward": 0.7138671875, "reward_std": 15.415876388549805, "rewards/rm_reward_func/mean": 0.7138671875, "rewards/rm_reward_func/std": 21.385021209716797, "step": 558 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 348.0, "completions/mean_terminated_length": 249.60000610351562, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "epoch": 0.4472, "grad_norm": 16.69782829284668, "kl": 9.87890625, "learning_rate": 1e-06, "loss": 0.6521, "num_tokens": 7518204.0, "reward": 0.142578125, "reward_std": 15.55780029296875, "rewards/rm_reward_func/mean": 0.142578125, "rewards/rm_reward_func/std": 18.24248695373535, "step": 559 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 486.0, "completions/mean_length": 212.71875, "completions/mean_terminated_length": 169.96429443359375, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "epoch": 0.448, "grad_norm": 96.58130645751953, "kl": 6.24658203125, "learning_rate": 1e-06, "loss": 0.5224, "num_tokens": 7527819.0, "reward": 1.61529541015625, "reward_std": 13.240062713623047, "rewards/rm_reward_func/mean": 1.61529541015625, "rewards/rm_reward_func/std": 14.038684844970703, "step": 560 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 475.0, "completions/mean_length": 239.09375, "completions/mean_terminated_length": 210.86207580566406, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "epoch": 0.4488, "grad_norm": 44.059059143066406, "kl": 11.875, "learning_rate": 1e-06, "loss": 0.4508, "num_tokens": 7538014.0, "reward": -12.23046875, "reward_std": 13.231216430664062, "rewards/rm_reward_func/mean": -12.23046875, "rewards/rm_reward_func/std": 17.87043571472168, "step": 561 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 441.0, "completions/mean_length": 162.28125, "completions/mean_terminated_length": 126.10344696044922, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "epoch": 0.4496, "grad_norm": 23.391313552856445, "kl": 6.7119140625, "learning_rate": 1e-06, "loss": 0.3591, "num_tokens": 7546287.0, "reward": -9.3154296875, "reward_std": 6.772148132324219, "rewards/rm_reward_func/mean": -9.3154296875, "rewards/rm_reward_func/std": 11.018516540527344, "step": 562 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 479.0, "completions/mean_length": 366.1875, "completions/mean_terminated_length": 317.5833435058594, "completions/min_length": 153.0, "completions/min_terminated_length": 153.0, "epoch": 0.4504, "grad_norm": 41.109046936035156, "kl": 7.4140625, "learning_rate": 1e-06, "loss": 0.3541, "num_tokens": 7560213.0, "reward": -8.3702392578125, "reward_std": 10.981157302856445, "rewards/rm_reward_func/mean": -8.3702392578125, "rewards/rm_reward_func/std": 13.88586711883545, "step": 563 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 339.5, "completions/mean_terminated_length": 291.1999816894531, "completions/min_length": 23.0, "completions/min_terminated_length": 23.0, "epoch": 0.4512, "grad_norm": 34.89735412597656, "kl": 4.27685546875, "learning_rate": 1e-06, "loss": 0.0249, "num_tokens": 7574949.0, "reward": 1.029541015625, "reward_std": 12.96914291381836, "rewards/rm_reward_func/mean": 1.029541015625, "rewards/rm_reward_func/std": 22.506608963012695, "step": 564 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 475.0, "completions/mean_length": 272.375, "completions/mean_terminated_length": 217.07693481445312, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "epoch": 0.452, "grad_norm": 10.21695613861084, "kl": 5.6484375, "learning_rate": 1e-06, "loss": 0.3193, "num_tokens": 7586177.0, "reward": -7.305419921875, "reward_std": 8.39257526397705, "rewards/rm_reward_func/mean": -7.305419921875, "rewards/rm_reward_func/std": 10.208649635314941, "step": 565 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 196.34375, "completions/mean_terminated_length": 163.6896514892578, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "epoch": 0.4528, "grad_norm": 74.78276062011719, "kl": 3.7001953125, "learning_rate": 1e-06, "loss": -0.0352, "num_tokens": 7595588.0, "reward": -7.03125, "reward_std": 6.5332746505737305, "rewards/rm_reward_func/mean": -7.03125, "rewards/rm_reward_func/std": 15.433568000793457, "step": 566 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 303.4375, "completions/mean_terminated_length": 289.5333557128906, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "epoch": 0.4536, "grad_norm": 212.21656799316406, "kl": 2.8095703125, "learning_rate": 1e-06, "loss": -0.0082, "num_tokens": 7607330.0, "reward": -1.4860076904296875, "reward_std": 13.193567276000977, "rewards/rm_reward_func/mean": -1.4860076904296875, "rewards/rm_reward_func/std": 17.839399337768555, "step": 567 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 236.78125, "completions/mean_terminated_length": 173.2692413330078, "completions/min_length": 20.0, "completions/min_terminated_length": 20.0, "epoch": 0.4544, "grad_norm": 22.839019775390625, "kl": 4.947509765625, "learning_rate": 1e-06, "loss": 0.3036, "num_tokens": 7619147.0, "reward": -4.39599609375, "reward_std": 10.989629745483398, "rewards/rm_reward_func/mean": -4.39599609375, "rewards/rm_reward_func/std": 14.866265296936035, "step": 568 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 261.25, "completions/mean_terminated_length": 253.16128540039062, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "epoch": 0.4552, "grad_norm": 10.727757453918457, "kl": 2.3701171875, "learning_rate": 1e-06, "loss": -0.1091, "num_tokens": 7629659.0, "reward": 2.4486083984375, "reward_std": 13.246370315551758, "rewards/rm_reward_func/mean": 2.4486083984375, "rewards/rm_reward_func/std": 20.726552963256836, "step": 569 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 495.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 218.90625, "completions/mean_terminated_length": 218.90625, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "epoch": 0.456, "grad_norm": 12.797094345092773, "kl": 2.15869140625, "learning_rate": 1e-06, "loss": 0.3033, "num_tokens": 7641048.0, "reward": 14.328125, "reward_std": 6.0160746574401855, "rewards/rm_reward_func/mean": 14.328125, "rewards/rm_reward_func/std": 12.738174438476562, "step": 570 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 443.0, "completions/mean_length": 239.65625, "completions/mean_terminated_length": 189.22222900390625, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "epoch": 0.4568, "grad_norm": 28.028533935546875, "kl": 6.15625, "learning_rate": 1e-06, "loss": 0.1429, "num_tokens": 7651101.0, "reward": -17.66259765625, "reward_std": 4.737618446350098, "rewards/rm_reward_func/mean": -17.66259765625, "rewards/rm_reward_func/std": 7.526523590087891, "step": 571 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 493.0, "completions/mean_length": 244.9375, "completions/mean_terminated_length": 183.3076934814453, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "epoch": 0.4576, "grad_norm": 28.566667556762695, "kl": 4.779541015625, "learning_rate": 1e-06, "loss": 0.31, "num_tokens": 7661483.0, "reward": -11.293212890625, "reward_std": 4.30594539642334, "rewards/rm_reward_func/mean": -11.293212890625, "rewards/rm_reward_func/std": 9.476552963256836, "step": 572 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 372.5, "completions/mean_terminated_length": 299.4285888671875, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "epoch": 0.4584, "grad_norm": 9.447068214416504, "kl": 3.060302734375, "learning_rate": 1e-06, "loss": 0.061, "num_tokens": 7679483.0, "reward": -9.2685546875, "reward_std": 7.054027080535889, "rewards/rm_reward_func/mean": -9.2685546875, "rewards/rm_reward_func/std": 12.366097450256348, "step": 573 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 337.125, "completions/mean_terminated_length": 304.7407531738281, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "epoch": 0.4592, "grad_norm": 11.237780570983887, "kl": 4.16357421875, "learning_rate": 1e-06, "loss": 0.166, "num_tokens": 7692759.0, "reward": -8.243682861328125, "reward_std": 9.03434944152832, "rewards/rm_reward_func/mean": -8.243682861328125, "rewards/rm_reward_func/std": 10.684294700622559, "step": 574 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 259.28125, "completions/mean_terminated_length": 160.3913116455078, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "epoch": 0.46, "grad_norm": 26.71776008605957, "kl": 2.917724609375, "learning_rate": 1e-06, "loss": 0.0122, "num_tokens": 7704504.0, "reward": -0.0777587890625, "reward_std": 7.303743839263916, "rewards/rm_reward_func/mean": -0.0777587890625, "rewards/rm_reward_func/std": 13.445627212524414, "step": 575 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 478.0, "completions/mean_length": 329.5625, "completions/mean_terminated_length": 278.47998046875, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "epoch": 0.4608, "grad_norm": 29.287355422973633, "kl": 3.5146484375, "learning_rate": 1e-06, "loss": 0.1701, "num_tokens": 7718426.0, "reward": -9.46527099609375, "reward_std": 5.940766334533691, "rewards/rm_reward_func/mean": -9.46527099609375, "rewards/rm_reward_func/std": 8.79275131225586, "step": 576 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 271.375, "completions/mean_terminated_length": 226.8148193359375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "epoch": 0.4616, "grad_norm": 31.825645446777344, "kl": 1.86328125, "learning_rate": 1e-06, "loss": -0.0067, "num_tokens": 7731526.0, "reward": 7.0263671875, "reward_std": 8.148383140563965, "rewards/rm_reward_func/mean": 7.0263671875, "rewards/rm_reward_func/std": 15.413996696472168, "step": 577 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 470.0, "completions/mean_length": 321.90625, "completions/mean_terminated_length": 302.2413635253906, "completions/min_length": 94.0, "completions/min_terminated_length": 94.0, "epoch": 0.4624, "grad_norm": 9.535029411315918, "kl": 3.79296875, "learning_rate": 1e-06, "loss": 0.0741, "num_tokens": 7746387.0, "reward": -4.50048828125, "reward_std": 10.548614501953125, "rewards/rm_reward_func/mean": -4.50048828125, "rewards/rm_reward_func/std": 18.594907760620117, "step": 578 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 345.0, "completions/mean_length": 201.5, "completions/mean_terminated_length": 98.0, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "epoch": 0.4632, "grad_norm": 8.471213340759277, "kl": 2.734375, "learning_rate": 1e-06, "loss": -0.1795, "num_tokens": 7755707.0, "reward": -7.9576416015625, "reward_std": 9.555851936340332, "rewards/rm_reward_func/mean": -7.9576416015625, "rewards/rm_reward_func/std": 11.002127647399902, "step": 579 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 479.0, "completions/mean_length": 220.96875, "completions/mean_terminated_length": 190.86207580566406, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "epoch": 0.464, "grad_norm": 8.996735572814941, "kl": 1.244140625, "learning_rate": 1e-06, "loss": -0.0699, "num_tokens": 7765434.0, "reward": 3.6036376953125, "reward_std": 8.55752944946289, "rewards/rm_reward_func/mean": 3.6036376953125, "rewards/rm_reward_func/std": 18.499204635620117, "step": 580 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 346.0, "completions/mean_terminated_length": 281.0434875488281, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "epoch": 0.4648, "grad_norm": 40.83759689331055, "kl": 0.305419921875, "learning_rate": 1e-06, "loss": 0.1, "num_tokens": 7778234.0, "reward": 4.37225341796875, "reward_std": 7.295747756958008, "rewards/rm_reward_func/mean": 4.37225341796875, "rewards/rm_reward_func/std": 8.913907051086426, "step": 581 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 456.0, "completions/mean_length": 305.96875, "completions/mean_terminated_length": 237.2916717529297, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "epoch": 0.4656, "grad_norm": 37.06733703613281, "kl": 2.0625, "learning_rate": 1e-06, "loss": -0.0452, "num_tokens": 7791017.0, "reward": -5.02978515625, "reward_std": 5.757181167602539, "rewards/rm_reward_func/mean": -5.02978515625, "rewards/rm_reward_func/std": 9.750814437866211, "step": 582 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 491.0, "completions/mean_length": 238.90625, "completions/mean_terminated_length": 199.8928680419922, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "epoch": 0.4664, "grad_norm": 34.94654846191406, "kl": 1.6552734375, "learning_rate": 1e-06, "loss": -0.1114, "num_tokens": 7803718.0, "reward": 1.8824462890625, "reward_std": 10.042098999023438, "rewards/rm_reward_func/mean": 1.8824462890625, "rewards/rm_reward_func/std": 14.9249906539917, "step": 583 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 289.09375, "completions/mean_terminated_length": 257.25, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "epoch": 0.4672, "grad_norm": 4.869449138641357, "kl": 0.384765625, "learning_rate": 1e-06, "loss": -0.0789, "num_tokens": 7815969.0, "reward": 8.9342041015625, "reward_std": 9.483150482177734, "rewards/rm_reward_func/mean": 8.9342041015625, "rewards/rm_reward_func/std": 18.35944938659668, "step": 584 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.65625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 462.03125, "completions/mean_terminated_length": 366.6363830566406, "completions/min_length": 149.0, "completions/min_terminated_length": 149.0, "epoch": 0.468, "grad_norm": 5.1822285652160645, "kl": 0.37060546875, "learning_rate": 1e-06, "loss": -0.0113, "num_tokens": 7833826.0, "reward": 7.9248046875, "reward_std": 7.081142425537109, "rewards/rm_reward_func/mean": 7.9248046875, "rewards/rm_reward_func/std": 15.302535057067871, "step": 585 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 304.0, "completions/mean_length": 131.4375, "completions/mean_terminated_length": 119.16128540039062, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "epoch": 0.4688, "grad_norm": 18.64679718017578, "kl": 3.083984375, "learning_rate": 1e-06, "loss": 0.2691, "num_tokens": 7841144.0, "reward": 5.0889892578125, "reward_std": 6.059321880340576, "rewards/rm_reward_func/mean": 5.0889892578125, "rewards/rm_reward_func/std": 9.212634086608887, "step": 586 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 335.84375, "completions/mean_terminated_length": 310.6785888671875, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "epoch": 0.4696, "grad_norm": 8.921401023864746, "kl": 2.075439453125, "learning_rate": 1e-06, "loss": -0.1186, "num_tokens": 7853947.0, "reward": -0.1695556640625, "reward_std": 9.398602485656738, "rewards/rm_reward_func/mean": -0.1695556640625, "rewards/rm_reward_func/std": 15.199664115905762, "step": 587 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 491.0, "completions/mean_length": 301.5, "completions/mean_terminated_length": 271.4285888671875, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "epoch": 0.4704, "grad_norm": 49.62898635864258, "kl": 0.86962890625, "learning_rate": 1e-06, "loss": -0.058, "num_tokens": 7865547.0, "reward": -1.8533935546875, "reward_std": 7.218017578125, "rewards/rm_reward_func/mean": -1.8533935546875, "rewards/rm_reward_func/std": 9.405008316040039, "step": 588 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 503.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 333.375, "completions/mean_terminated_length": 333.375, "completions/min_length": 171.0, "completions/min_terminated_length": 171.0, "epoch": 0.4712, "grad_norm": 15.543201446533203, "kl": 0.32958984375, "learning_rate": 1e-06, "loss": -0.0359, "num_tokens": 7879767.0, "reward": -5.37200927734375, "reward_std": 3.4672133922576904, "rewards/rm_reward_func/mean": -5.37200927734375, "rewards/rm_reward_func/std": 7.8242692947387695, "step": 589 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 377.84375, "completions/mean_terminated_length": 340.2799987792969, "completions/min_length": 94.0, "completions/min_terminated_length": 94.0, "epoch": 0.472, "grad_norm": 7.237336158752441, "kl": 0.4638671875, "learning_rate": 1e-06, "loss": -0.0329, "num_tokens": 7894162.0, "reward": 6.933837890625, "reward_std": 7.349064826965332, "rewards/rm_reward_func/mean": 6.933837890625, "rewards/rm_reward_func/std": 12.662103652954102, "step": 590 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 270.09375, "completions/mean_terminated_length": 245.0689697265625, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "epoch": 0.4728, "grad_norm": 9.466930389404297, "kl": 1.3511962890625, "learning_rate": 1e-06, "loss": 0.1344, "num_tokens": 7905197.0, "reward": -5.54095458984375, "reward_std": 5.511334419250488, "rewards/rm_reward_func/mean": -5.54095458984375, "rewards/rm_reward_func/std": 9.380617141723633, "step": 591 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 187.5625, "completions/mean_terminated_length": 127.48148345947266, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "epoch": 0.4736, "grad_norm": 10.78067684173584, "kl": 0.6640625, "learning_rate": 1e-06, "loss": -0.0794, "num_tokens": 7913271.0, "reward": 4.76513671875, "reward_std": 7.13428258895874, "rewards/rm_reward_func/mean": 4.76513671875, "rewards/rm_reward_func/std": 9.546843528747559, "step": 592 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 297.8125, "completions/mean_terminated_length": 214.0, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "epoch": 0.4744, "grad_norm": 5.899567604064941, "kl": 0.46875, "learning_rate": 1e-06, "loss": -0.1007, "num_tokens": 7926073.0, "reward": 1.1571044921875, "reward_std": 7.337251663208008, "rewards/rm_reward_func/mean": 1.1571044921875, "rewards/rm_reward_func/std": 10.56173038482666, "step": 593 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.4375, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 446.40625, "completions/mean_terminated_length": 395.3888854980469, "completions/min_length": 261.0, "completions/min_terminated_length": 261.0, "epoch": 0.4752, "grad_norm": 41.35441589355469, "kl": 2.393310546875, "learning_rate": 1e-06, "loss": 0.12, "num_tokens": 7942310.0, "reward": -0.016571044921875, "reward_std": 9.10225772857666, "rewards/rm_reward_func/mean": -0.016571044921875, "rewards/rm_reward_func/std": 14.169840812683105, "step": 594 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 468.0, "completions/mean_length": 234.15625, "completions/mean_terminated_length": 215.6333465576172, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "epoch": 0.476, "grad_norm": 13.656573295593262, "kl": 6.201904296875, "learning_rate": 1e-06, "loss": 0.7573, "num_tokens": 7954699.0, "reward": 6.20806884765625, "reward_std": 9.577255249023438, "rewards/rm_reward_func/mean": 6.20806884765625, "rewards/rm_reward_func/std": 14.503737449645996, "step": 595 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 455.46875, "completions/mean_terminated_length": 398.9375, "completions/min_length": 202.0, "completions/min_terminated_length": 202.0, "epoch": 0.4768, "grad_norm": 5.957422256469727, "kl": 2.294677734375, "learning_rate": 1e-06, "loss": 0.071, "num_tokens": 7973066.0, "reward": -7.954833984375, "reward_std": 6.558610916137695, "rewards/rm_reward_func/mean": -7.954833984375, "rewards/rm_reward_func/std": 9.106958389282227, "step": 596 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 506.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 269.96875, "completions/mean_terminated_length": 269.96875, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "epoch": 0.4776, "grad_norm": 8.821357727050781, "kl": 0.6884765625, "learning_rate": 1e-06, "loss": 0.0222, "num_tokens": 7984641.0, "reward": 6.18316650390625, "reward_std": 5.323254585266113, "rewards/rm_reward_func/mean": 6.18316650390625, "rewards/rm_reward_func/std": 14.378352165222168, "step": 597 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 231.8125, "completions/mean_terminated_length": 222.77418518066406, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "epoch": 0.4784, "grad_norm": 34.12923812866211, "kl": 7.343505859375, "learning_rate": 1e-06, "loss": 0.3296, "num_tokens": 7995091.0, "reward": 2.2025146484375, "reward_std": 7.2968525886535645, "rewards/rm_reward_func/mean": 2.2025146484375, "rewards/rm_reward_func/std": 13.586483001708984, "step": 598 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 302.03125, "completions/mean_terminated_length": 280.3103332519531, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "epoch": 0.4792, "grad_norm": 6.906167507171631, "kl": 0.790283203125, "learning_rate": 1e-06, "loss": -0.1121, "num_tokens": 8007900.0, "reward": 4.49267578125, "reward_std": 6.870022296905518, "rewards/rm_reward_func/mean": 4.49267578125, "rewards/rm_reward_func/std": 9.993231773376465, "step": 599 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 474.0, "completions/mean_length": 248.53125, "completions/mean_terminated_length": 210.8928680419922, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "epoch": 0.48, "grad_norm": 31.199399948120117, "kl": 5.575927734375, "learning_rate": 1e-06, "loss": 0.4506, "num_tokens": 8022373.0, "reward": -0.75750732421875, "reward_std": 8.437265396118164, "rewards/rm_reward_func/mean": -0.75750732421875, "rewards/rm_reward_func/std": 14.56766128540039, "step": 600 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 446.0, "completions/mean_length": 286.75, "completions/mean_terminated_length": 234.7692413330078, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "epoch": 0.4808, "grad_norm": 61.780311584472656, "kl": 6.131103515625, "learning_rate": 1e-06, "loss": 0.1509, "num_tokens": 8034205.0, "reward": 8.10595703125, "reward_std": 10.480352401733398, "rewards/rm_reward_func/mean": 8.10595703125, "rewards/rm_reward_func/std": 14.536432266235352, "step": 601 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 477.0, "completions/mean_length": 346.59375, "completions/mean_terminated_length": 259.952392578125, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "epoch": 0.4816, "grad_norm": 213.77684020996094, "kl": 21.95849609375, "learning_rate": 1e-06, "loss": 1.2326, "num_tokens": 8048464.0, "reward": -6.7283935546875, "reward_std": 6.977231979370117, "rewards/rm_reward_func/mean": -6.7283935546875, "rewards/rm_reward_func/std": 14.559057235717773, "step": 602 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 491.0, "completions/mean_length": 360.71875, "completions/mean_terminated_length": 318.3599853515625, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "epoch": 0.4824, "grad_norm": 13.767047882080078, "kl": 8.62158203125, "learning_rate": 1e-06, "loss": 0.4078, "num_tokens": 8063103.0, "reward": -1.645751953125, "reward_std": 12.11515998840332, "rewards/rm_reward_func/mean": -1.645751953125, "rewards/rm_reward_func/std": 15.272088050842285, "step": 603 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 481.0, "completions/mean_length": 322.0, "completions/mean_terminated_length": 278.15386962890625, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "epoch": 0.4832, "grad_norm": 24.88362693786621, "kl": 8.8076171875, "learning_rate": 1e-06, "loss": 0.2726, "num_tokens": 8075391.0, "reward": -6.1571044921875, "reward_std": 18.25802230834961, "rewards/rm_reward_func/mean": -6.1571044921875, "rewards/rm_reward_func/std": 19.966968536376953, "step": 604 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 453.0, "completions/mean_length": 218.71875, "completions/mean_terminated_length": 176.82144165039062, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "epoch": 0.484, "grad_norm": 11.096504211425781, "kl": 3.28759765625, "learning_rate": 1e-06, "loss": 0.169, "num_tokens": 8086822.0, "reward": 6.87744140625, "reward_std": 7.8228960037231445, "rewards/rm_reward_func/mean": 6.87744140625, "rewards/rm_reward_func/std": 12.922216415405273, "step": 605 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 420.03125, "completions/mean_terminated_length": 398.8077087402344, "completions/min_length": 201.0, "completions/min_terminated_length": 201.0, "epoch": 0.4848, "grad_norm": 10.59483528137207, "kl": 3.473876953125, "learning_rate": 1e-06, "loss": 0.0768, "num_tokens": 8106231.0, "reward": 15.840576171875, "reward_std": 15.33732795715332, "rewards/rm_reward_func/mean": 15.840576171875, "rewards/rm_reward_func/std": 20.74690818786621, "step": 606 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 429.15625, "completions/mean_terminated_length": 385.76190185546875, "completions/min_length": 167.0, "completions/min_terminated_length": 167.0, "epoch": 0.4856, "grad_norm": 9.831181526184082, "kl": 4.218505859375, "learning_rate": 1e-06, "loss": 0.1415, "num_tokens": 8124388.0, "reward": -1.52099609375, "reward_std": 13.66183853149414, "rewards/rm_reward_func/mean": -1.52099609375, "rewards/rm_reward_func/std": 18.234188079833984, "step": 607 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 358.75, "completions/mean_terminated_length": 323.3846130371094, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "epoch": 0.4864, "grad_norm": 10.595935821533203, "kl": 2.94921875, "learning_rate": 1e-06, "loss": -0.0267, "num_tokens": 8138108.0, "reward": 5.33203125, "reward_std": 11.863018035888672, "rewards/rm_reward_func/mean": 5.33203125, "rewards/rm_reward_func/std": 24.679624557495117, "step": 608 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 460.0, "completions/mean_length": 217.8125, "completions/mean_terminated_length": 208.32257080078125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "epoch": 0.4872, "grad_norm": 22.514549255371094, "kl": 5.7451171875, "learning_rate": 1e-06, "loss": 0.3125, "num_tokens": 8150726.0, "reward": 0.619384765625, "reward_std": 10.594083786010742, "rewards/rm_reward_func/mean": 0.619384765625, "rewards/rm_reward_func/std": 15.584603309631348, "step": 609 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 392.0, "completions/mean_length": 236.28125, "completions/mean_terminated_length": 207.7586212158203, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "epoch": 0.488, "grad_norm": 25.03614616394043, "kl": 8.80859375, "learning_rate": 1e-06, "loss": 0.7636, "num_tokens": 8162679.0, "reward": 4.9461517333984375, "reward_std": 10.481914520263672, "rewards/rm_reward_func/mean": 4.9461517333984375, "rewards/rm_reward_func/std": 14.789458274841309, "step": 610 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 405.28125, "completions/mean_terminated_length": 363.5217590332031, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "epoch": 0.4888, "grad_norm": 9.863654136657715, "kl": 4.548828125, "learning_rate": 1e-06, "loss": 0.1808, "num_tokens": 8177880.0, "reward": 1.1624755859375, "reward_std": 12.410948753356934, "rewards/rm_reward_func/mean": 1.1624755859375, "rewards/rm_reward_func/std": 17.47865867614746, "step": 611 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 382.3125, "completions/mean_terminated_length": 358.2962951660156, "completions/min_length": 162.0, "completions/min_terminated_length": 162.0, "epoch": 0.4896, "grad_norm": 16.073301315307617, "kl": 3.01123046875, "learning_rate": 1e-06, "loss": 0.0967, "num_tokens": 8192650.0, "reward": 0.911651611328125, "reward_std": 8.299646377563477, "rewards/rm_reward_func/mean": 0.911651611328125, "rewards/rm_reward_func/std": 14.285723686218262, "step": 612 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 450.0, "completions/mean_length": 341.65625, "completions/mean_terminated_length": 275.0, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "epoch": 0.4904, "grad_norm": 37.6219596862793, "kl": 9.759765625, "learning_rate": 1e-06, "loss": 0.5727, "num_tokens": 8205711.0, "reward": -1.9541015625, "reward_std": 9.328568458557129, "rewards/rm_reward_func/mean": -1.9541015625, "rewards/rm_reward_func/std": 21.1943302154541, "step": 613 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 474.0, "completions/mean_length": 267.03125, "completions/mean_terminated_length": 185.375, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "epoch": 0.4912, "grad_norm": 153.5972900390625, "kl": 7.2421875, "learning_rate": 1e-06, "loss": 0.3504, "num_tokens": 8218824.0, "reward": -9.59063720703125, "reward_std": 9.781373023986816, "rewards/rm_reward_func/mean": -9.59063720703125, "rewards/rm_reward_func/std": 11.414146423339844, "step": 614 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 492.0, "completions/mean_length": 188.53125, "completions/mean_terminated_length": 166.9666748046875, "completions/min_length": 12.0, "completions/min_terminated_length": 12.0, "epoch": 0.492, "grad_norm": 28.16669464111328, "kl": 5.5712890625, "learning_rate": 1e-06, "loss": 0.1814, "num_tokens": 8228137.0, "reward": -3.1532211303710938, "reward_std": 12.097888946533203, "rewards/rm_reward_func/mean": -3.1532211303710938, "rewards/rm_reward_func/std": 20.61545181274414, "step": 615 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 313.53125, "completions/mean_terminated_length": 235.86956787109375, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "epoch": 0.4928, "grad_norm": 40.259830474853516, "kl": 10.94140625, "learning_rate": 1e-06, "loss": 0.5348, "num_tokens": 8241418.0, "reward": -5.6474609375, "reward_std": 11.151725769042969, "rewards/rm_reward_func/mean": -5.6474609375, "rewards/rm_reward_func/std": 13.004457473754883, "step": 616 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 265.28125, "completions/mean_terminated_length": 230.0357208251953, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "epoch": 0.4936, "grad_norm": 9.916903495788574, "kl": 5.4296875, "learning_rate": 1e-06, "loss": 0.1988, "num_tokens": 8251955.0, "reward": -0.9298095703125, "reward_std": 15.462509155273438, "rewards/rm_reward_func/mean": -0.9298095703125, "rewards/rm_reward_func/std": 16.872419357299805, "step": 617 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 471.0, "completions/mean_length": 204.09375, "completions/mean_terminated_length": 172.2413787841797, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "epoch": 0.4944, "grad_norm": 63.2779655456543, "kl": 8.203125, "learning_rate": 1e-06, "loss": 0.2805, "num_tokens": 8264246.0, "reward": -4.11572265625, "reward_std": 13.06856918334961, "rewards/rm_reward_func/mean": -4.11572265625, "rewards/rm_reward_func/std": 14.47628116607666, "step": 618 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 448.0, "completions/mean_length": 346.78125, "completions/mean_terminated_length": 247.65000915527344, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "epoch": 0.4952, "grad_norm": 44.88002014160156, "kl": 9.136474609375, "learning_rate": 1e-06, "loss": 0.5063, "num_tokens": 8277079.0, "reward": -9.4381103515625, "reward_std": 5.765872955322266, "rewards/rm_reward_func/mean": -9.4381103515625, "rewards/rm_reward_func/std": 11.600118637084961, "step": 619 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 480.0, "completions/mean_length": 293.15625, "completions/mean_terminated_length": 270.5172424316406, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "epoch": 0.496, "grad_norm": 37.431556701660156, "kl": 2.302734375, "learning_rate": 1e-06, "loss": 0.0574, "num_tokens": 8288420.0, "reward": -2.6103515625, "reward_std": 13.910895347595215, "rewards/rm_reward_func/mean": -2.6103515625, "rewards/rm_reward_func/std": 16.85130500793457, "step": 620 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 473.0, "completions/mean_length": 258.84375, "completions/mean_terminated_length": 222.6785888671875, "completions/min_length": 97.0, "completions/min_terminated_length": 97.0, "epoch": 0.4968, "grad_norm": 13.83285903930664, "kl": 7.1640625, "learning_rate": 1e-06, "loss": 0.2555, "num_tokens": 8298487.0, "reward": -6.125762939453125, "reward_std": 8.449478149414062, "rewards/rm_reward_func/mean": -6.125762939453125, "rewards/rm_reward_func/std": 9.483611106872559, "step": 621 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 435.0, "completions/mean_length": 191.71875, "completions/mean_terminated_length": 170.36666870117188, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "epoch": 0.4976, "grad_norm": 12.294360160827637, "kl": 5.13671875, "learning_rate": 1e-06, "loss": 0.2798, "num_tokens": 8310150.0, "reward": 7.74169921875, "reward_std": 11.26545238494873, "rewards/rm_reward_func/mean": 7.74169921875, "rewards/rm_reward_func/std": 15.538458824157715, "step": 622 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 461.0, "completions/mean_length": 234.125, "completions/mean_terminated_length": 170.0, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "epoch": 0.4984, "grad_norm": 7.7052130699157715, "kl": 2.001953125, "learning_rate": 1e-06, "loss": 0.0997, "num_tokens": 8321794.0, "reward": 6.1068115234375, "reward_std": 6.466403961181641, "rewards/rm_reward_func/mean": 6.1068115234375, "rewards/rm_reward_func/std": 10.666815757751465, "step": 623 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 271.15625, "completions/mean_terminated_length": 226.55555725097656, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "epoch": 0.4992, "grad_norm": 37.59298324584961, "kl": 6.85546875, "learning_rate": 1e-06, "loss": 0.4438, "num_tokens": 8333623.0, "reward": -11.0943603515625, "reward_std": 6.584903717041016, "rewards/rm_reward_func/mean": -11.0943603515625, "rewards/rm_reward_func/std": 11.544700622558594, "step": 624 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 454.0, "completions/mean_length": 282.40625, "completions/mean_terminated_length": 205.875, "completions/min_length": 8.0, "completions/min_terminated_length": 8.0, "epoch": 0.5, "grad_norm": 50.945106506347656, "kl": 7.80419921875, "learning_rate": 1e-06, "loss": 0.1727, "num_tokens": 8346988.0, "reward": -10.51025390625, "reward_std": 4.640988349914551, "rewards/rm_reward_func/mean": -10.51025390625, "rewards/rm_reward_func/std": 19.07770538330078, "step": 625 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 276.5625, "completions/mean_terminated_length": 260.8666687011719, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "epoch": 0.5008, "grad_norm": 8.402812957763672, "kl": 3.9267578125, "learning_rate": 1e-06, "loss": 0.0878, "num_tokens": 8358310.0, "reward": 10.74951171875, "reward_std": 11.159141540527344, "rewards/rm_reward_func/mean": 10.74951171875, "rewards/rm_reward_func/std": 24.454740524291992, "step": 626 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 314.59375, "completions/mean_terminated_length": 294.17242431640625, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "epoch": 0.5016, "grad_norm": 7.214862823486328, "kl": 3.716796875, "learning_rate": 1e-06, "loss": 0.1032, "num_tokens": 8371249.0, "reward": -1.5455322265625, "reward_std": 13.805252075195312, "rewards/rm_reward_func/mean": -1.5455322265625, "rewards/rm_reward_func/std": 15.483540534973145, "step": 627 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 311.90625, "completions/mean_terminated_length": 255.87998962402344, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "epoch": 0.5024, "grad_norm": 5.6593217849731445, "kl": 2.5107421875, "learning_rate": 1e-06, "loss": -0.0017, "num_tokens": 8383638.0, "reward": 5.8603515625, "reward_std": 11.585649490356445, "rewards/rm_reward_func/mean": 5.8603515625, "rewards/rm_reward_func/std": 14.266550064086914, "step": 628 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 272.90625, "completions/mean_terminated_length": 238.75001525878906, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "epoch": 0.5032, "grad_norm": 11.517411231994629, "kl": 6.453125, "learning_rate": 1e-06, "loss": 0.2772, "num_tokens": 8394563.0, "reward": -4.296142578125, "reward_std": 15.2503023147583, "rewards/rm_reward_func/mean": -4.296142578125, "rewards/rm_reward_func/std": 15.96473217010498, "step": 629 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 310.53125, "completions/mean_terminated_length": 281.75, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "epoch": 0.504, "grad_norm": 6.986127853393555, "kl": 1.2666015625, "learning_rate": 1e-06, "loss": -0.0682, "num_tokens": 8410244.0, "reward": 1.4921875, "reward_std": 4.761030673980713, "rewards/rm_reward_func/mean": 1.4921875, "rewards/rm_reward_func/std": 14.25729751586914, "step": 630 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 302.84375, "completions/mean_terminated_length": 288.9000244140625, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "epoch": 0.5048, "grad_norm": 7.896935939788818, "kl": 4.4833984375, "learning_rate": 1e-06, "loss": 0.0858, "num_tokens": 8424207.0, "reward": 5.5654296875, "reward_std": 8.17580509185791, "rewards/rm_reward_func/mean": 5.5654296875, "rewards/rm_reward_func/std": 20.969762802124023, "step": 631 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 271.1875, "completions/mean_terminated_length": 246.27586364746094, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "epoch": 0.5056, "grad_norm": 10.68520450592041, "kl": 1.8544921875, "learning_rate": 1e-06, "loss": 0.1494, "num_tokens": 8436013.0, "reward": 1.7723541259765625, "reward_std": 9.819549560546875, "rewards/rm_reward_func/mean": 1.7723541259765625, "rewards/rm_reward_func/std": 15.363565444946289, "step": 632 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 327.875, "completions/mean_terminated_length": 315.6000061035156, "completions/min_length": 115.0, "completions/min_terminated_length": 115.0, "epoch": 0.5064, "grad_norm": 8.58014965057373, "kl": 1.8515625, "learning_rate": 1e-06, "loss": 0.1069, "num_tokens": 8449793.0, "reward": 6.67364501953125, "reward_std": 11.383968353271484, "rewards/rm_reward_func/mean": 6.67364501953125, "rewards/rm_reward_func/std": 13.357749938964844, "step": 633 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 450.0, "completions/mean_length": 272.25, "completions/mean_terminated_length": 205.1199951171875, "completions/min_length": 99.0, "completions/min_terminated_length": 99.0, "epoch": 0.5072, "grad_norm": 9.852457046508789, "kl": 2.81201171875, "learning_rate": 1e-06, "loss": 0.135, "num_tokens": 8460897.0, "reward": -4.5697021484375, "reward_std": 5.767214775085449, "rewards/rm_reward_func/mean": -4.5697021484375, "rewards/rm_reward_func/std": 9.804206848144531, "step": 634 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 277.375, "completions/mean_terminated_length": 253.10345458984375, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "epoch": 0.508, "grad_norm": 13.91521167755127, "kl": 4.9462890625, "learning_rate": 1e-06, "loss": 0.1343, "num_tokens": 8472789.0, "reward": 0.547698974609375, "reward_std": 12.389558792114258, "rewards/rm_reward_func/mean": 0.547698974609375, "rewards/rm_reward_func/std": 21.48679542541504, "step": 635 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 246.15625, "completions/mean_terminated_length": 196.92593383789062, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "epoch": 0.5088, "grad_norm": 10.53650188446045, "kl": 2.0712890625, "learning_rate": 1e-06, "loss": 0.1688, "num_tokens": 8484914.0, "reward": -1.3759765625, "reward_std": 2.3449654579162598, "rewards/rm_reward_func/mean": -1.3759765625, "rewards/rm_reward_func/std": 12.400309562683105, "step": 636 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 333.875, "completions/mean_terminated_length": 264.1739196777344, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "epoch": 0.5096, "grad_norm": 24.877025604248047, "kl": 6.34228515625, "learning_rate": 1e-06, "loss": 0.2245, "num_tokens": 8500902.0, "reward": -9.4365234375, "reward_std": 3.848327159881592, "rewards/rm_reward_func/mean": -9.4365234375, "rewards/rm_reward_func/std": 13.026070594787598, "step": 637 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 410.40625, "completions/mean_terminated_length": 364.227294921875, "completions/min_length": 92.0, "completions/min_terminated_length": 92.0, "epoch": 0.5104, "grad_norm": 8.95622730255127, "kl": 1.86279296875, "learning_rate": 1e-06, "loss": -0.0112, "num_tokens": 8515907.0, "reward": 1.7091064453125, "reward_std": 10.488420486450195, "rewards/rm_reward_func/mean": 1.7091064453125, "rewards/rm_reward_func/std": 15.33460807800293, "step": 638 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 274.0625, "completions/mean_terminated_length": 207.44000244140625, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "epoch": 0.5112, "grad_norm": 13.682821273803711, "kl": 5.75537109375, "learning_rate": 1e-06, "loss": 0.0707, "num_tokens": 8528813.0, "reward": -12.8310546875, "reward_std": 8.058960914611816, "rewards/rm_reward_func/mean": -12.8310546875, "rewards/rm_reward_func/std": 10.044072151184082, "step": 639 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 252.125, "completions/mean_terminated_length": 215.00001525878906, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "epoch": 0.512, "grad_norm": 10.187383651733398, "kl": 3.99658203125, "learning_rate": 1e-06, "loss": 0.1334, "num_tokens": 8539441.0, "reward": 1.024169921875, "reward_std": 7.485830783843994, "rewards/rm_reward_func/mean": 1.024169921875, "rewards/rm_reward_func/std": 11.878993034362793, "step": 640 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 197.96875, "completions/mean_terminated_length": 139.8148193359375, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "epoch": 0.5128, "grad_norm": 6.773480415344238, "kl": 4.6123046875, "learning_rate": 1e-06, "loss": 0.2292, "num_tokens": 8549136.0, "reward": -1.02978515625, "reward_std": 5.95222282409668, "rewards/rm_reward_func/mean": -1.02978515625, "rewards/rm_reward_func/std": 9.468921661376953, "step": 641 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 480.0, "completions/mean_length": 213.53125, "completions/mean_terminated_length": 170.8928680419922, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "epoch": 0.5136, "grad_norm": 6.013530254364014, "kl": 0.498046875, "learning_rate": 1e-06, "loss": 0.0151, "num_tokens": 8560257.0, "reward": 8.07672119140625, "reward_std": 5.212985038757324, "rewards/rm_reward_func/mean": 8.07672119140625, "rewards/rm_reward_func/std": 9.585315704345703, "step": 642 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 419.125, "completions/mean_terminated_length": 393.1199951171875, "completions/min_length": 233.0, "completions/min_terminated_length": 233.0, "epoch": 0.5144, "grad_norm": 6.766341686248779, "kl": 1.09765625, "learning_rate": 1e-06, "loss": 0.0954, "num_tokens": 8575533.0, "reward": 14.398681640625, "reward_std": 10.005803108215332, "rewards/rm_reward_func/mean": 14.398681640625, "rewards/rm_reward_func/std": 12.649359703063965, "step": 643 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 221.21875, "completions/mean_terminated_length": 167.37037658691406, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "epoch": 0.5152, "grad_norm": 15.727855682373047, "kl": 0.90576171875, "learning_rate": 1e-06, "loss": -0.0352, "num_tokens": 8587596.0, "reward": 7.632568359375, "reward_std": 3.306748867034912, "rewards/rm_reward_func/mean": 7.632568359375, "rewards/rm_reward_func/std": 9.895013809204102, "step": 644 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 465.0, "completions/mean_length": 252.40625, "completions/mean_terminated_length": 235.10000610351562, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "epoch": 0.516, "grad_norm": 20.60538673400879, "kl": 1.96630859375, "learning_rate": 1e-06, "loss": -0.0003, "num_tokens": 8599505.0, "reward": 5.3416748046875, "reward_std": 11.507960319519043, "rewards/rm_reward_func/mean": 5.3416748046875, "rewards/rm_reward_func/std": 17.1043758392334, "step": 645 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.46875, "completions/max_length": 512.0, "completions/max_terminated_length": 469.0, "completions/mean_length": 281.15625, "completions/mean_terminated_length": 77.47058868408203, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "epoch": 0.5168, "grad_norm": 10.928936004638672, "kl": 1.356201171875, "learning_rate": 1e-06, "loss": 0.0367, "num_tokens": 8610998.0, "reward": -1.3593597412109375, "reward_std": 3.8079264163970947, "rewards/rm_reward_func/mean": -1.3593597412109375, "rewards/rm_reward_func/std": 8.6522216796875, "step": 646 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 486.0, "completions/mean_length": 362.59375, "completions/mean_terminated_length": 312.79168701171875, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "epoch": 0.5176, "grad_norm": 7.342215538024902, "kl": 2.76806640625, "learning_rate": 1e-06, "loss": 0.1712, "num_tokens": 8625361.0, "reward": -2.8857421875, "reward_std": 11.833799362182617, "rewards/rm_reward_func/mean": -2.8857421875, "rewards/rm_reward_func/std": 15.419763565063477, "step": 647 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 392.90625, "completions/mean_terminated_length": 321.45001220703125, "completions/min_length": 172.0, "completions/min_terminated_length": 172.0, "epoch": 0.5184, "grad_norm": 16.402130126953125, "kl": 4.22412109375, "learning_rate": 1e-06, "loss": 0.1936, "num_tokens": 8639934.0, "reward": 0.00299072265625, "reward_std": 5.563260078430176, "rewards/rm_reward_func/mean": 0.00299072265625, "rewards/rm_reward_func/std": 17.13874053955078, "step": 648 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 484.0, "completions/mean_length": 174.40625, "completions/mean_terminated_length": 163.51612854003906, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "epoch": 0.5192, "grad_norm": 8.888167381286621, "kl": 0.8353271484375, "learning_rate": 1e-06, "loss": -0.0946, "num_tokens": 8652491.0, "reward": 0.316680908203125, "reward_std": 4.313920021057129, "rewards/rm_reward_func/mean": 0.316680908203125, "rewards/rm_reward_func/std": 10.923198699951172, "step": 649 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 491.0, "completions/mean_length": 290.03125, "completions/mean_terminated_length": 189.13636779785156, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "epoch": 0.52, "grad_norm": 6.159624099731445, "kl": 1.846435546875, "learning_rate": 1e-06, "loss": 0.007, "num_tokens": 8665452.0, "reward": 4.464111328125, "reward_std": 7.780835151672363, "rewards/rm_reward_func/mean": 4.464111328125, "rewards/rm_reward_func/std": 10.548760414123535, "step": 650 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 265.6875, "completions/mean_terminated_length": 208.84616088867188, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "epoch": 0.5208, "grad_norm": 11.240917205810547, "kl": 1.3095703125, "learning_rate": 1e-06, "loss": 0.063, "num_tokens": 8676746.0, "reward": 2.96875, "reward_std": 5.190629005432129, "rewards/rm_reward_func/mean": 2.96875, "rewards/rm_reward_func/std": 12.547528266906738, "step": 651 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.4375, "completions/max_length": 512.0, "completions/max_terminated_length": 453.0, "completions/mean_length": 299.84375, "completions/mean_terminated_length": 134.8333282470703, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "epoch": 0.5216, "grad_norm": 9.644030570983887, "kl": 0.29296875, "learning_rate": 1e-06, "loss": -0.0588, "num_tokens": 8689357.0, "reward": 0.384521484375, "reward_std": 3.9550719261169434, "rewards/rm_reward_func/mean": 0.384521484375, "rewards/rm_reward_func/std": 12.727936744689941, "step": 652 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 347.9375, "completions/mean_terminated_length": 317.5555725097656, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "epoch": 0.5224, "grad_norm": 10.710493087768555, "kl": 0.42578125, "learning_rate": 1e-06, "loss": 0.0054, "num_tokens": 8702963.0, "reward": 10.232421875, "reward_std": 6.635782241821289, "rewards/rm_reward_func/mean": 10.232421875, "rewards/rm_reward_func/std": 9.094783782958984, "step": 653 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 470.0, "completions/mean_length": 320.875, "completions/mean_terminated_length": 301.10345458984375, "completions/min_length": 153.0, "completions/min_terminated_length": 153.0, "epoch": 0.5232, "grad_norm": 6.3006792068481445, "kl": 0.252197265625, "learning_rate": 1e-06, "loss": 0.0086, "num_tokens": 8716151.0, "reward": 8.225341796875, "reward_std": 7.599928855895996, "rewards/rm_reward_func/mean": 8.225341796875, "rewards/rm_reward_func/std": 19.037904739379883, "step": 654 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 375.875, "completions/mean_terminated_length": 294.20001220703125, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "epoch": 0.524, "grad_norm": 8.997413635253906, "kl": 0.244140625, "learning_rate": 1e-06, "loss": 0.0777, "num_tokens": 8732867.0, "reward": 4.292236328125, "reward_std": 7.305292129516602, "rewards/rm_reward_func/mean": 4.292236328125, "rewards/rm_reward_func/std": 11.710954666137695, "step": 655 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 357.53125, "completions/mean_terminated_length": 297.08697509765625, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "epoch": 0.5248, "grad_norm": 6.956607818603516, "kl": 0.75, "learning_rate": 1e-06, "loss": 0.0366, "num_tokens": 8747420.0, "reward": 11.81103515625, "reward_std": 6.258151054382324, "rewards/rm_reward_func/mean": 11.81103515625, "rewards/rm_reward_func/std": 15.444485664367676, "step": 656 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 507.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 360.75, "completions/mean_terminated_length": 360.75, "completions/min_length": 148.0, "completions/min_terminated_length": 148.0, "epoch": 0.5256, "grad_norm": 6.72127103805542, "kl": 0.552490234375, "learning_rate": 1e-06, "loss": 0.0153, "num_tokens": 8760988.0, "reward": 10.35595703125, "reward_std": 5.374138832092285, "rewards/rm_reward_func/mean": 10.35595703125, "rewards/rm_reward_func/std": 6.271159648895264, "step": 657 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 286.84375, "completions/mean_terminated_length": 223.79998779296875, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "epoch": 0.5264, "grad_norm": 7.279812335968018, "kl": 0.7529296875, "learning_rate": 1e-06, "loss": 0.0554, "num_tokens": 8772311.0, "reward": 3.248779296875, "reward_std": 4.097195625305176, "rewards/rm_reward_func/mean": 3.248779296875, "rewards/rm_reward_func/std": 11.073434829711914, "step": 658 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.46875, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 444.15625, "completions/mean_terminated_length": 384.29412841796875, "completions/min_length": 264.0, "completions/min_terminated_length": 264.0, "epoch": 0.5272, "grad_norm": 7.7773356437683105, "kl": 0.760986328125, "learning_rate": 1e-06, "loss": -0.0174, "num_tokens": 8788708.0, "reward": -2.1377792358398438, "reward_std": 4.090878486633301, "rewards/rm_reward_func/mean": -2.1377792358398438, "rewards/rm_reward_func/std": 10.053132057189941, "step": 659 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 417.59375, "completions/mean_terminated_length": 391.1600036621094, "completions/min_length": 282.0, "completions/min_terminated_length": 282.0, "epoch": 0.528, "grad_norm": 6.207853317260742, "kl": 0.521240234375, "learning_rate": 1e-06, "loss": 0.0471, "num_tokens": 8804327.0, "reward": 3.51171875, "reward_std": 6.670271873474121, "rewards/rm_reward_func/mean": 3.51171875, "rewards/rm_reward_func/std": 13.217809677124023, "step": 660 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 222.25, "completions/mean_terminated_length": 141.1199951171875, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "epoch": 0.5288, "grad_norm": 3.570781946182251, "kl": 2.3623046875, "learning_rate": 1e-06, "loss": 0.0811, "num_tokens": 8815471.0, "reward": 3.49066162109375, "reward_std": 3.8420963287353516, "rewards/rm_reward_func/mean": 3.49066162109375, "rewards/rm_reward_func/std": 9.580389976501465, "step": 661 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 459.0, "completions/mean_length": 334.34375, "completions/mean_terminated_length": 275.125, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "epoch": 0.5296, "grad_norm": 16.329147338867188, "kl": 0.8173828125, "learning_rate": 1e-06, "loss": 0.0685, "num_tokens": 8828386.0, "reward": 1.876220703125, "reward_std": 7.676386833190918, "rewards/rm_reward_func/mean": 1.876220703125, "rewards/rm_reward_func/std": 12.502883911132812, "step": 662 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 398.625, "completions/mean_terminated_length": 330.6000061035156, "completions/min_length": 90.0, "completions/min_terminated_length": 90.0, "epoch": 0.5304, "grad_norm": 6.071362018585205, "kl": 1.19775390625, "learning_rate": 1e-06, "loss": 0.1563, "num_tokens": 8844302.0, "reward": 1.69091796875, "reward_std": 8.782705307006836, "rewards/rm_reward_func/mean": 1.69091796875, "rewards/rm_reward_func/std": 12.565754890441895, "step": 663 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 232.6875, "completions/mean_terminated_length": 223.6774139404297, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "epoch": 0.5312, "grad_norm": 11.870681762695312, "kl": 4.712890625, "learning_rate": 1e-06, "loss": 0.2744, "num_tokens": 8855052.0, "reward": -1.134765625, "reward_std": 4.481464385986328, "rewards/rm_reward_func/mean": -1.134765625, "rewards/rm_reward_func/std": 17.394426345825195, "step": 664 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 358.21875, "completions/mean_terminated_length": 353.258056640625, "completions/min_length": 195.0, "completions/min_terminated_length": 195.0, "epoch": 0.532, "grad_norm": 6.449156284332275, "kl": 2.161865234375, "learning_rate": 1e-06, "loss": 0.0442, "num_tokens": 8868555.0, "reward": 2.504180908203125, "reward_std": 8.275325775146484, "rewards/rm_reward_func/mean": 2.504180908203125, "rewards/rm_reward_func/std": 15.419231414794922, "step": 665 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 412.40625, "completions/mean_terminated_length": 373.4347839355469, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "epoch": 0.5328, "grad_norm": 15.121233940124512, "kl": 4.3203125, "learning_rate": 1e-06, "loss": 0.0551, "num_tokens": 8886344.0, "reward": 3.4033203125, "reward_std": 12.214666366577148, "rewards/rm_reward_func/mean": 3.4033203125, "rewards/rm_reward_func/std": 17.266735076904297, "step": 666 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 277.125, "completions/mean_terminated_length": 222.92308044433594, "completions/min_length": 92.0, "completions/min_terminated_length": 92.0, "epoch": 0.5336, "grad_norm": 20.014434814453125, "kl": 5.343505859375, "learning_rate": 1e-06, "loss": 0.1802, "num_tokens": 8900948.0, "reward": -5.2738800048828125, "reward_std": 6.429419994354248, "rewards/rm_reward_func/mean": -5.2738800048828125, "rewards/rm_reward_func/std": 11.191460609436035, "step": 667 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 286.21875, "completions/mean_terminated_length": 223.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "epoch": 0.5344, "grad_norm": 46.5359001159668, "kl": 16.328125, "learning_rate": 1e-06, "loss": 0.6287, "num_tokens": 8912667.0, "reward": -19.37646484375, "reward_std": 8.034518241882324, "rewards/rm_reward_func/mean": -19.37646484375, "rewards/rm_reward_func/std": 11.764345169067383, "step": 668 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 374.09375, "completions/mean_terminated_length": 354.39288330078125, "completions/min_length": 143.0, "completions/min_terminated_length": 143.0, "epoch": 0.5352, "grad_norm": 8.606017112731934, "kl": 4.453125, "learning_rate": 1e-06, "loss": 0.1118, "num_tokens": 8926590.0, "reward": 5.37371826171875, "reward_std": 12.284038543701172, "rewards/rm_reward_func/mean": 5.37371826171875, "rewards/rm_reward_func/std": 20.101335525512695, "step": 669 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 341.5625, "completions/mean_terminated_length": 323.9310302734375, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "epoch": 0.536, "grad_norm": 20.169708251953125, "kl": 2.861328125, "learning_rate": 1e-06, "loss": 0.3412, "num_tokens": 8941640.0, "reward": 8.84033203125, "reward_std": 7.4539666175842285, "rewards/rm_reward_func/mean": 8.84033203125, "rewards/rm_reward_func/std": 11.035738945007324, "step": 670 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 315.84375, "completions/mean_terminated_length": 250.45834350585938, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "epoch": 0.5368, "grad_norm": 8.888484001159668, "kl": 1.6533203125, "learning_rate": 1e-06, "loss": -0.0764, "num_tokens": 8955819.0, "reward": 1.8431396484375, "reward_std": 9.139184951782227, "rewards/rm_reward_func/mean": 1.8431396484375, "rewards/rm_reward_func/std": 12.700114250183105, "step": 671 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 273.5625, "completions/mean_terminated_length": 265.8709716796875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "epoch": 0.5376, "grad_norm": 10.354080200195312, "kl": 1.771484375, "learning_rate": 1e-06, "loss": 0.1823, "num_tokens": 8966917.0, "reward": 3.476806640625, "reward_std": 4.19705867767334, "rewards/rm_reward_func/mean": 3.476806640625, "rewards/rm_reward_func/std": 17.924257278442383, "step": 672 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 331.0, "completions/mean_length": 100.0, "completions/mean_terminated_length": 86.70967102050781, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "epoch": 0.5384, "grad_norm": 28.584192276000977, "kl": 4.08642578125, "learning_rate": 1e-06, "loss": 0.4196, "num_tokens": 8974605.0, "reward": 1.3251953125, "reward_std": 7.180523872375488, "rewards/rm_reward_func/mean": 1.3251953125, "rewards/rm_reward_func/std": 14.241349220275879, "step": 673 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 406.0, "completions/mean_length": 226.21875, "completions/mean_terminated_length": 130.95834350585938, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "epoch": 0.5392, "grad_norm": 35.00379943847656, "kl": 4.3994140625, "learning_rate": 1e-06, "loss": 0.1642, "num_tokens": 8986524.0, "reward": -5.6318359375, "reward_std": 3.752939224243164, "rewards/rm_reward_func/mean": -5.6318359375, "rewards/rm_reward_func/std": 12.199244499206543, "step": 674 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 456.0, "completions/mean_length": 313.5625, "completions/mean_terminated_length": 223.3636474609375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "epoch": 0.54, "grad_norm": 41.50772476196289, "kl": 8.8515625, "learning_rate": 1e-06, "loss": 0.2941, "num_tokens": 8999078.0, "reward": -8.677978515625, "reward_std": 4.428235054016113, "rewards/rm_reward_func/mean": -8.677978515625, "rewards/rm_reward_func/std": 14.32275104522705, "step": 675 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 366.0, "completions/mean_length": 183.375, "completions/mean_terminated_length": 122.51851654052734, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "epoch": 0.5408, "grad_norm": 7.0692219734191895, "kl": 4.337890625, "learning_rate": 1e-06, "loss": 0.1818, "num_tokens": 9007402.0, "reward": -1.382568359375, "reward_std": 6.2474846839904785, "rewards/rm_reward_func/mean": -1.382568359375, "rewards/rm_reward_func/std": 9.534759521484375, "step": 676 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 376.6875, "completions/mean_terminated_length": 357.3571472167969, "completions/min_length": 101.0, "completions/min_terminated_length": 101.0, "epoch": 0.5416, "grad_norm": 52.67711639404297, "kl": 6.74609375, "learning_rate": 1e-06, "loss": 0.187, "num_tokens": 9021656.0, "reward": 0.373046875, "reward_std": 13.075346946716309, "rewards/rm_reward_func/mean": 0.373046875, "rewards/rm_reward_func/std": 22.373414993286133, "step": 677 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40625, "completions/max_length": 512.0, "completions/max_terminated_length": 492.0, "completions/mean_length": 340.0625, "completions/mean_terminated_length": 222.42105102539062, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "epoch": 0.5424, "grad_norm": 8.724742889404297, "kl": 2.99169921875, "learning_rate": 1e-06, "loss": 0.0752, "num_tokens": 9037954.0, "reward": 3.04296875, "reward_std": 12.51136589050293, "rewards/rm_reward_func/mean": 3.04296875, "rewards/rm_reward_func/std": 18.065021514892578, "step": 678 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 190.28125, "completions/mean_terminated_length": 144.32144165039062, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "epoch": 0.5432, "grad_norm": 17.94440269470215, "kl": 4.4853515625, "learning_rate": 1e-06, "loss": 0.4292, "num_tokens": 9046651.0, "reward": 0.7139892578125, "reward_std": 3.5457606315612793, "rewards/rm_reward_func/mean": 0.7139892578125, "rewards/rm_reward_func/std": 9.750553131103516, "step": 679 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 474.0, "completions/mean_length": 312.09375, "completions/mean_terminated_length": 298.7666931152344, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "epoch": 0.544, "grad_norm": 21.662830352783203, "kl": 7.321533203125, "learning_rate": 1e-06, "loss": 0.1919, "num_tokens": 9059910.0, "reward": -3.553955078125, "reward_std": 10.071688652038574, "rewards/rm_reward_func/mean": -3.553955078125, "rewards/rm_reward_func/std": 14.8821382522583, "step": 680 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 266.25, "completions/mean_terminated_length": 220.74073791503906, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "epoch": 0.5448, "grad_norm": 6.978208065032959, "kl": 4.47412109375, "learning_rate": 1e-06, "loss": 0.1302, "num_tokens": 9070510.0, "reward": -1.48779296875, "reward_std": 9.718238830566406, "rewards/rm_reward_func/mean": -1.48779296875, "rewards/rm_reward_func/std": 15.631418228149414, "step": 681 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 252.6875, "completions/mean_terminated_length": 215.6428680419922, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "epoch": 0.5456, "grad_norm": 132.93753051757812, "kl": 9.15625, "learning_rate": 1e-06, "loss": 0.2597, "num_tokens": 9080252.0, "reward": -10.84326171875, "reward_std": 9.151596069335938, "rewards/rm_reward_func/mean": -10.84326171875, "rewards/rm_reward_func/std": 14.070874214172363, "step": 682 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 434.0, "completions/mean_length": 145.9375, "completions/mean_terminated_length": 121.53334045410156, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "epoch": 0.5464, "grad_norm": 22.839073181152344, "kl": 5.984375, "learning_rate": 1e-06, "loss": 0.1499, "num_tokens": 9088962.0, "reward": -9.11358642578125, "reward_std": 12.002496719360352, "rewards/rm_reward_func/mean": -9.11358642578125, "rewards/rm_reward_func/std": 16.55123519897461, "step": 683 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 475.0, "completions/mean_length": 238.875, "completions/mean_terminated_length": 210.6206817626953, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "epoch": 0.5472, "grad_norm": 37.41764831542969, "kl": 5.50390625, "learning_rate": 1e-06, "loss": 0.0492, "num_tokens": 9098534.0, "reward": -3.8291015625, "reward_std": 8.362621307373047, "rewards/rm_reward_func/mean": -3.8291015625, "rewards/rm_reward_func/std": 15.430756568908691, "step": 684 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 464.0, "completions/mean_length": 230.5625, "completions/mean_terminated_length": 221.48387145996094, "completions/min_length": 23.0, "completions/min_terminated_length": 23.0, "epoch": 0.548, "grad_norm": 19.959774017333984, "kl": 5.6953125, "learning_rate": 1e-06, "loss": 0.0816, "num_tokens": 9108744.0, "reward": -6.86328125, "reward_std": 15.26154613494873, "rewards/rm_reward_func/mean": -6.86328125, "rewards/rm_reward_func/std": 18.045209884643555, "step": 685 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 371.625, "completions/mean_terminated_length": 345.629638671875, "completions/min_length": 109.0, "completions/min_terminated_length": 109.0, "epoch": 0.5488, "grad_norm": 10.441740989685059, "kl": 4.70703125, "learning_rate": 1e-06, "loss": 0.2121, "num_tokens": 9123308.0, "reward": 0.5693359375, "reward_std": 7.536800384521484, "rewards/rm_reward_func/mean": 0.5693359375, "rewards/rm_reward_func/std": 17.190340042114258, "step": 686 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 184.9375, "completions/mean_terminated_length": 151.10345458984375, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "epoch": 0.5496, "grad_norm": 12.261507987976074, "kl": 2.86083984375, "learning_rate": 1e-06, "loss": 0.0786, "num_tokens": 9132018.0, "reward": 5.0924072265625, "reward_std": 3.7133607864379883, "rewards/rm_reward_func/mean": 5.0924072265625, "rewards/rm_reward_func/std": 8.903958320617676, "step": 687 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 493.0, "completions/max_terminated_length": 493.0, "completions/mean_length": 254.96875, "completions/mean_terminated_length": 254.96875, "completions/min_length": 8.0, "completions/min_terminated_length": 8.0, "epoch": 0.5504, "grad_norm": 11.390403747558594, "kl": 3.8935546875, "learning_rate": 1e-06, "loss": -0.1216, "num_tokens": 9143177.0, "reward": -8.523193359375, "reward_std": 11.859602928161621, "rewards/rm_reward_func/mean": -8.523193359375, "rewards/rm_reward_func/std": 12.442675590515137, "step": 688 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 407.0, "completions/mean_length": 254.90625, "completions/mean_terminated_length": 237.7666778564453, "completions/min_length": 19.0, "completions/min_terminated_length": 19.0, "epoch": 0.5512, "grad_norm": 9.756564140319824, "kl": 4.615234375, "learning_rate": 1e-06, "loss": 0.1961, "num_tokens": 9155374.0, "reward": -8.00146484375, "reward_std": 10.193147659301758, "rewards/rm_reward_func/mean": -8.00146484375, "rewards/rm_reward_func/std": 11.081028938293457, "step": 689 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 344.90625, "completions/mean_terminated_length": 279.5217590332031, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "epoch": 0.552, "grad_norm": 12.181934356689453, "kl": 3.39208984375, "learning_rate": 1e-06, "loss": 0.0899, "num_tokens": 9170707.0, "reward": -6.829345703125, "reward_std": 5.473711013793945, "rewards/rm_reward_func/mean": -6.829345703125, "rewards/rm_reward_func/std": 10.798709869384766, "step": 690 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 484.0, "completions/mean_length": 332.0625, "completions/mean_terminated_length": 320.0666809082031, "completions/min_length": 89.0, "completions/min_terminated_length": 89.0, "epoch": 0.5528, "grad_norm": 10.006911277770996, "kl": 1.677978515625, "learning_rate": 1e-06, "loss": -0.0336, "num_tokens": 9184165.0, "reward": 3.05224609375, "reward_std": 7.134219169616699, "rewards/rm_reward_func/mean": 3.05224609375, "rewards/rm_reward_func/std": 13.809707641601562, "step": 691 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 485.0, "completions/mean_length": 250.5625, "completions/mean_terminated_length": 202.1481475830078, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "epoch": 0.5536, "grad_norm": 6.326221942901611, "kl": 0.51953125, "learning_rate": 1e-06, "loss": -0.0631, "num_tokens": 9195959.0, "reward": 12.587890625, "reward_std": 4.309545040130615, "rewards/rm_reward_func/mean": 12.587890625, "rewards/rm_reward_func/std": 6.763890266418457, "step": 692 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 474.0, "completions/mean_length": 217.6875, "completions/mean_terminated_length": 208.19354248046875, "completions/min_length": 12.0, "completions/min_terminated_length": 12.0, "epoch": 0.5544, "grad_norm": 15.060230255126953, "kl": 2.18994140625, "learning_rate": 1e-06, "loss": 0.1014, "num_tokens": 9206501.0, "reward": -7.5948486328125, "reward_std": 6.410397529602051, "rewards/rm_reward_func/mean": -7.5948486328125, "rewards/rm_reward_func/std": 9.005189895629883, "step": 693 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 444.0, "completions/mean_length": 206.0, "completions/mean_terminated_length": 196.1290283203125, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "epoch": 0.5552, "grad_norm": 12.398327827453613, "kl": 1.31103515625, "learning_rate": 1e-06, "loss": -0.1765, "num_tokens": 9218037.0, "reward": -7.2659912109375, "reward_std": 6.091501712799072, "rewards/rm_reward_func/mean": -7.2659912109375, "rewards/rm_reward_func/std": 11.163793563842773, "step": 694 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 345.84375, "completions/mean_terminated_length": 322.1071472167969, "completions/min_length": 155.0, "completions/min_terminated_length": 155.0, "epoch": 0.556, "grad_norm": 6.776838779449463, "kl": 2.171875, "learning_rate": 1e-06, "loss": 0.0345, "num_tokens": 9231384.0, "reward": -5.34033203125, "reward_std": 5.828690052032471, "rewards/rm_reward_func/mean": -5.34033203125, "rewards/rm_reward_func/std": 6.941495418548584, "step": 695 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 266.84375, "completions/mean_terminated_length": 221.44444274902344, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "epoch": 0.5568, "grad_norm": 6.661186695098877, "kl": 0.462158203125, "learning_rate": 1e-06, "loss": 0.0817, "num_tokens": 9244219.0, "reward": 3.683349609375, "reward_std": 6.5144147872924805, "rewards/rm_reward_func/mean": 3.683349609375, "rewards/rm_reward_func/std": 10.859166145324707, "step": 696 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 407.0, "completions/mean_terminated_length": 382.7692565917969, "completions/min_length": 139.0, "completions/min_terminated_length": 139.0, "epoch": 0.5576, "grad_norm": 16.05687141418457, "kl": 0.77197265625, "learning_rate": 1e-06, "loss": -0.0237, "num_tokens": 9260139.0, "reward": 3.64599609375, "reward_std": 6.319300651550293, "rewards/rm_reward_func/mean": 3.64599609375, "rewards/rm_reward_func/std": 14.248506546020508, "step": 697 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 474.0, "completions/mean_length": 395.09375, "completions/mean_terminated_length": 349.34783935546875, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "epoch": 0.5584, "grad_norm": 7.179235458374023, "kl": 1.25146484375, "learning_rate": 1e-06, "loss": -0.0007, "num_tokens": 9275854.0, "reward": 3.7479248046875, "reward_std": 7.614123821258545, "rewards/rm_reward_func/mean": 3.7479248046875, "rewards/rm_reward_func/std": 17.118871688842773, "step": 698 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 320.0, "completions/mean_length": 225.90625, "completions/mean_terminated_length": 172.92593383789062, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "epoch": 0.5592, "grad_norm": 17.386568069458008, "kl": 0.8544921875, "learning_rate": 1e-06, "loss": -0.137, "num_tokens": 9286667.0, "reward": -11.1297607421875, "reward_std": 6.347601890563965, "rewards/rm_reward_func/mean": -11.1297607421875, "rewards/rm_reward_func/std": 6.936679363250732, "step": 699 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 433.0, "completions/max_terminated_length": 433.0, "completions/mean_length": 210.59375, "completions/mean_terminated_length": 210.59375, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "epoch": 0.56, "grad_norm": 13.88813591003418, "kl": 1.533203125, "learning_rate": 1e-06, "loss": 0.082, "num_tokens": 9296486.0, "reward": -1.9166259765625, "reward_std": 6.144941806793213, "rewards/rm_reward_func/mean": -1.9166259765625, "rewards/rm_reward_func/std": 11.042118072509766, "step": 700 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 417.0, "completions/mean_length": 235.3125, "completions/mean_terminated_length": 216.86668395996094, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "epoch": 0.5608, "grad_norm": 14.585112571716309, "kl": 1.34814453125, "learning_rate": 1e-06, "loss": -0.0184, "num_tokens": 9306088.0, "reward": -1.9686279296875, "reward_std": 6.677350997924805, "rewards/rm_reward_func/mean": -1.9686279296875, "rewards/rm_reward_func/std": 8.160258293151855, "step": 701 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 452.0, "completions/mean_length": 243.34375, "completions/mean_terminated_length": 225.433349609375, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "epoch": 0.5616, "grad_norm": 16.384502410888672, "kl": 1.71630859375, "learning_rate": 1e-06, "loss": 0.2464, "num_tokens": 9319155.0, "reward": 0.691162109375, "reward_std": 6.911756992340088, "rewards/rm_reward_func/mean": 0.691162109375, "rewards/rm_reward_func/std": 17.84705924987793, "step": 702 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 382.0, "completions/max_terminated_length": 382.0, "completions/mean_length": 148.78125, "completions/mean_terminated_length": 148.78125, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "epoch": 0.5624, "grad_norm": 6.983405113220215, "kl": 0.526123046875, "learning_rate": 1e-06, "loss": -0.1855, "num_tokens": 9326356.0, "reward": -2.7777099609375, "reward_std": 3.3739712238311768, "rewards/rm_reward_func/mean": -2.7777099609375, "rewards/rm_reward_func/std": 7.716010570526123, "step": 703 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 476.0, "completions/mean_length": 366.71875, "completions/mean_terminated_length": 357.0333557128906, "completions/min_length": 142.0, "completions/min_terminated_length": 142.0, "epoch": 0.5632, "grad_norm": 10.195340156555176, "kl": 1.0078125, "learning_rate": 1e-06, "loss": -0.0249, "num_tokens": 9340835.0, "reward": 5.092041015625, "reward_std": 8.344895362854004, "rewards/rm_reward_func/mean": 5.092041015625, "rewards/rm_reward_func/std": 13.963737487792969, "step": 704 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 430.0, "completions/mean_length": 267.59375, "completions/mean_terminated_length": 259.70965576171875, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "epoch": 0.564, "grad_norm": 15.919843673706055, "kl": 1.554931640625, "learning_rate": 1e-06, "loss": 0.159, "num_tokens": 9352078.0, "reward": -2.619140625, "reward_std": 7.604259490966797, "rewards/rm_reward_func/mean": -2.619140625, "rewards/rm_reward_func/std": 10.991021156311035, "step": 705 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 407.375, "completions/mean_terminated_length": 366.4347839355469, "completions/min_length": 95.0, "completions/min_terminated_length": 95.0, "epoch": 0.5648, "grad_norm": 12.950475692749023, "kl": 2.766845703125, "learning_rate": 1e-06, "loss": 0.068, "num_tokens": 9367250.0, "reward": 2.364013671875, "reward_std": 8.381866455078125, "rewards/rm_reward_func/mean": 2.364013671875, "rewards/rm_reward_func/std": 12.802971839904785, "step": 706 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 298.75, "completions/mean_terminated_length": 259.2592468261719, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "epoch": 0.5656, "grad_norm": 6.933451175689697, "kl": 0.818603515625, "learning_rate": 1e-06, "loss": 0.0176, "num_tokens": 9379698.0, "reward": 4.7874755859375, "reward_std": 7.199830532073975, "rewards/rm_reward_func/mean": 4.7874755859375, "rewards/rm_reward_func/std": 9.206077575683594, "step": 707 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 441.0, "completions/max_terminated_length": 441.0, "completions/mean_length": 232.8125, "completions/mean_terminated_length": 232.8125, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "epoch": 0.5664, "grad_norm": 6.320394992828369, "kl": 0.52197265625, "learning_rate": 1e-06, "loss": -0.1093, "num_tokens": 9389684.0, "reward": 13.0057373046875, "reward_std": 8.226985931396484, "rewards/rm_reward_func/mean": 13.0057373046875, "rewards/rm_reward_func/std": 11.692206382751465, "step": 708 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 468.0, "completions/mean_length": 290.46875, "completions/mean_terminated_length": 275.70001220703125, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "epoch": 0.5672, "grad_norm": 9.328205108642578, "kl": 2.07763671875, "learning_rate": 1e-06, "loss": 0.1139, "num_tokens": 9400939.0, "reward": 2.2265625, "reward_std": 8.862954139709473, "rewards/rm_reward_func/mean": 2.2265625, "rewards/rm_reward_func/std": 12.553840637207031, "step": 709 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 286.625, "completions/mean_terminated_length": 223.51998901367188, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "epoch": 0.568, "grad_norm": 12.409536361694336, "kl": 4.6474609375, "learning_rate": 1e-06, "loss": 0.2891, "num_tokens": 9412575.0, "reward": 3.2001953125, "reward_std": 6.288336753845215, "rewards/rm_reward_func/mean": 3.2001953125, "rewards/rm_reward_func/std": 10.550227165222168, "step": 710 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 338.90625, "completions/mean_terminated_length": 314.1785888671875, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "epoch": 0.5688, "grad_norm": 5.891862869262695, "kl": 1.07666015625, "learning_rate": 1e-06, "loss": 0.0507, "num_tokens": 9427844.0, "reward": 7.5908203125, "reward_std": 3.023407459259033, "rewards/rm_reward_func/mean": 7.5908203125, "rewards/rm_reward_func/std": 12.348587036132812, "step": 711 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 338.59375, "completions/mean_terminated_length": 259.7727355957031, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "epoch": 0.5696, "grad_norm": 5.630682468414307, "kl": 2.202880859375, "learning_rate": 1e-06, "loss": 0.0399, "num_tokens": 9441263.0, "reward": 4.940185546875, "reward_std": 8.606001853942871, "rewards/rm_reward_func/mean": 4.940185546875, "rewards/rm_reward_func/std": 15.963677406311035, "step": 712 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.59375, "completions/max_length": 512.0, "completions/max_terminated_length": 475.0, "completions/mean_length": 381.65625, "completions/mean_terminated_length": 191.1538543701172, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "epoch": 0.5704, "grad_norm": 21.119258880615234, "kl": 10.06396484375, "learning_rate": 1e-06, "loss": 0.3982, "num_tokens": 9456948.0, "reward": -4.12841796875, "reward_std": 10.323966026306152, "rewards/rm_reward_func/mean": -4.12841796875, "rewards/rm_reward_func/std": 16.72847557067871, "step": 713 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.4375, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 382.0625, "completions/mean_terminated_length": 281.0, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "epoch": 0.5712, "grad_norm": 7.876568794250488, "kl": 5.89892578125, "learning_rate": 1e-06, "loss": 0.1956, "num_tokens": 9471470.0, "reward": -1.296875, "reward_std": 11.687822341918945, "rewards/rm_reward_func/mean": -1.296875, "rewards/rm_reward_func/std": 17.725196838378906, "step": 714 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 301.6875, "completions/mean_terminated_length": 219.3913116455078, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "epoch": 0.572, "grad_norm": 16.49651527404785, "kl": 6.4150390625, "learning_rate": 1e-06, "loss": 0.2308, "num_tokens": 9486132.0, "reward": -4.572021484375, "reward_std": 6.965249061584473, "rewards/rm_reward_func/mean": -4.572021484375, "rewards/rm_reward_func/std": 12.301454544067383, "step": 715 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 493.0, "completions/mean_length": 383.8125, "completions/mean_terminated_length": 306.8999938964844, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "epoch": 0.5728, "grad_norm": 7.270954608917236, "kl": 4.827392578125, "learning_rate": 1e-06, "loss": 0.2095, "num_tokens": 9503854.0, "reward": 20.08734130859375, "reward_std": 11.994423866271973, "rewards/rm_reward_func/mean": 20.08734130859375, "rewards/rm_reward_func/std": 29.330028533935547, "step": 716 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 394.40625, "completions/mean_terminated_length": 361.47998046875, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "epoch": 0.5736, "grad_norm": 5.99873161315918, "kl": 5.353271484375, "learning_rate": 1e-06, "loss": 0.1456, "num_tokens": 9519059.0, "reward": 0.1048583984375, "reward_std": 15.651195526123047, "rewards/rm_reward_func/mean": 0.1048583984375, "rewards/rm_reward_func/std": 15.79875373840332, "step": 717 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 340.53125, "completions/mean_terminated_length": 300.9615478515625, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "epoch": 0.5744, "grad_norm": 10.691807746887207, "kl": 6.0263671875, "learning_rate": 1e-06, "loss": 0.1929, "num_tokens": 9532652.0, "reward": 3.922882080078125, "reward_std": 8.720820426940918, "rewards/rm_reward_func/mean": 3.922882080078125, "rewards/rm_reward_func/std": 20.64432716369629, "step": 718 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 475.0, "completions/mean_length": 238.09375, "completions/mean_terminated_length": 161.39999389648438, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "epoch": 0.5752, "grad_norm": 5.596479892730713, "kl": 2.0068359375, "learning_rate": 1e-06, "loss": 0.0605, "num_tokens": 9545783.0, "reward": 6.065826416015625, "reward_std": 3.9478113651275635, "rewards/rm_reward_func/mean": 6.065826416015625, "rewards/rm_reward_func/std": 7.903535842895508, "step": 719 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 393.5, "completions/mean_terminated_length": 381.2413635253906, "completions/min_length": 209.0, "completions/min_terminated_length": 209.0, "epoch": 0.576, "grad_norm": 7.530872344970703, "kl": 4.125, "learning_rate": 1e-06, "loss": 0.0967, "num_tokens": 9561407.0, "reward": 8.341751098632812, "reward_std": 13.022964477539062, "rewards/rm_reward_func/mean": 8.341751098632812, "rewards/rm_reward_func/std": 16.08083724975586, "step": 720 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 481.0, "completions/mean_length": 346.71875, "completions/mean_terminated_length": 260.1428527832031, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "epoch": 0.5768, "grad_norm": 12.534733772277832, "kl": 4.27685546875, "learning_rate": 1e-06, "loss": 0.1581, "num_tokens": 9574310.0, "reward": -3.28759765625, "reward_std": 5.179610252380371, "rewards/rm_reward_func/mean": -3.28759765625, "rewards/rm_reward_func/std": 12.57066822052002, "step": 721 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 365.28125, "completions/mean_terminated_length": 307.86956787109375, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "epoch": 0.5776, "grad_norm": 8.022375106811523, "kl": 1.400390625, "learning_rate": 1e-06, "loss": 0.0708, "num_tokens": 9589439.0, "reward": 5.868088722229004, "reward_std": 7.8726677894592285, "rewards/rm_reward_func/mean": 5.868088722229004, "rewards/rm_reward_func/std": 16.735122680664062, "step": 722 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 419.90625, "completions/mean_terminated_length": 406.7500305175781, "completions/min_length": 284.0, "completions/min_terminated_length": 284.0, "epoch": 0.5784, "grad_norm": 6.199563980102539, "kl": 1.158935546875, "learning_rate": 1e-06, "loss": 0.0303, "num_tokens": 9604860.0, "reward": 7.0489501953125, "reward_std": 6.722829818725586, "rewards/rm_reward_func/mean": 7.0489501953125, "rewards/rm_reward_func/std": 10.408683776855469, "step": 723 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 410.5625, "completions/mean_terminated_length": 364.4545593261719, "completions/min_length": 104.0, "completions/min_terminated_length": 104.0, "epoch": 0.5792, "grad_norm": 8.32470703125, "kl": 2.0205078125, "learning_rate": 1e-06, "loss": 0.0497, "num_tokens": 9620638.0, "reward": 8.731201171875, "reward_std": 13.764381408691406, "rewards/rm_reward_func/mean": 8.731201171875, "rewards/rm_reward_func/std": 16.106142044067383, "step": 724 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 333.1875, "completions/mean_terminated_length": 239.52381896972656, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "epoch": 0.58, "grad_norm": 9.582738876342773, "kl": 2.6728515625, "learning_rate": 1e-06, "loss": 0.1381, "num_tokens": 9633836.0, "reward": 0.17033767700195312, "reward_std": 8.874990463256836, "rewards/rm_reward_func/mean": 0.17033767700195312, "rewards/rm_reward_func/std": 9.867877960205078, "step": 725 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 358.40625, "completions/mean_terminated_length": 277.952392578125, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "epoch": 0.5808, "grad_norm": 5.19415807723999, "kl": 2.4541015625, "learning_rate": 1e-06, "loss": 0.138, "num_tokens": 9647841.0, "reward": 3.939697265625, "reward_std": 5.236965179443359, "rewards/rm_reward_func/mean": 3.939697265625, "rewards/rm_reward_func/std": 12.523097038269043, "step": 726 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40625, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 385.40625, "completions/mean_terminated_length": 298.78948974609375, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "epoch": 0.5816, "grad_norm": 7.742301940917969, "kl": 2.68505859375, "learning_rate": 1e-06, "loss": 0.2645, "num_tokens": 9663086.0, "reward": 1.2738037109375, "reward_std": 6.155429363250732, "rewards/rm_reward_func/mean": 1.2738037109375, "rewards/rm_reward_func/std": 7.7377028465271, "step": 727 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 158.0, "completions/max_terminated_length": 158.0, "completions/mean_length": 79.90625, "completions/mean_terminated_length": 79.90625, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "epoch": 0.5824, "grad_norm": 7.791232585906982, "kl": 1.28955078125, "learning_rate": 1e-06, "loss": 0.0041, "num_tokens": 9670771.0, "reward": -0.384521484375, "reward_std": 2.3221030235290527, "rewards/rm_reward_func/mean": -0.384521484375, "rewards/rm_reward_func/std": 5.91448974609375, "step": 728 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40625, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 437.34375, "completions/mean_terminated_length": 386.2631530761719, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "epoch": 0.5832, "grad_norm": 6.162810325622559, "kl": 2.515625, "learning_rate": 1e-06, "loss": 0.0724, "num_tokens": 9687254.0, "reward": 4.739013671875, "reward_std": 13.861799240112305, "rewards/rm_reward_func/mean": 4.739013671875, "rewards/rm_reward_func/std": 15.850342750549316, "step": 729 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 445.0, "completions/mean_length": 365.15625, "completions/mean_terminated_length": 307.6956481933594, "completions/min_length": 105.0, "completions/min_terminated_length": 105.0, "epoch": 0.584, "grad_norm": 8.422097206115723, "kl": 3.58056640625, "learning_rate": 1e-06, "loss": 0.0715, "num_tokens": 9700803.0, "reward": -7.97607421875, "reward_std": 8.740215301513672, "rewards/rm_reward_func/mean": -7.97607421875, "rewards/rm_reward_func/std": 10.935211181640625, "step": 730 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 473.875, "completions/mean_terminated_length": 451.0, "completions/min_length": 347.0, "completions/min_terminated_length": 347.0, "epoch": 0.5848, "grad_norm": 9.889822006225586, "kl": 0.578125, "learning_rate": 1e-06, "loss": 0.0484, "num_tokens": 9718783.0, "reward": 15.586181640625, "reward_std": 11.35405445098877, "rewards/rm_reward_func/mean": 15.586181640625, "rewards/rm_reward_func/std": 11.300307273864746, "step": 731 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 479.0, "completions/mean_length": 292.59375, "completions/mean_terminated_length": 269.89654541015625, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "epoch": 0.5856, "grad_norm": 7.809492111206055, "kl": 0.862060546875, "learning_rate": 1e-06, "loss": 0.0524, "num_tokens": 9732722.0, "reward": 8.6011962890625, "reward_std": 3.8772120475769043, "rewards/rm_reward_func/mean": 8.6011962890625, "rewards/rm_reward_func/std": 7.693234920501709, "step": 732 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40625, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 441.21875, "completions/mean_terminated_length": 392.78948974609375, "completions/min_length": 259.0, "completions/min_terminated_length": 259.0, "epoch": 0.5864, "grad_norm": 7.425821781158447, "kl": 3.7294921875, "learning_rate": 1e-06, "loss": 0.1576, "num_tokens": 9749089.0, "reward": 9.70703125, "reward_std": 12.684297561645508, "rewards/rm_reward_func/mean": 9.70703125, "rewards/rm_reward_func/std": 14.209367752075195, "step": 733 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 470.0, "completions/mean_length": 310.65625, "completions/mean_terminated_length": 231.86956787109375, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "epoch": 0.5872, "grad_norm": 24.46310806274414, "kl": 5.77197265625, "learning_rate": 1e-06, "loss": 0.2606, "num_tokens": 9761590.0, "reward": -5.23046875, "reward_std": 6.721203804016113, "rewards/rm_reward_func/mean": -5.23046875, "rewards/rm_reward_func/std": 17.831560134887695, "step": 734 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 351.03125, "completions/mean_terminated_length": 288.0434875488281, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "epoch": 0.588, "grad_norm": 4.2976603507995605, "kl": 1.31982421875, "learning_rate": 1e-06, "loss": 0.0227, "num_tokens": 9775831.0, "reward": 10.992095947265625, "reward_std": 7.566171646118164, "rewards/rm_reward_func/mean": 10.992095947265625, "rewards/rm_reward_func/std": 9.95695686340332, "step": 735 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 261.46875, "completions/mean_terminated_length": 225.6785888671875, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "epoch": 0.5888, "grad_norm": 7.958250999450684, "kl": 2.50341796875, "learning_rate": 1e-06, "loss": 0.137, "num_tokens": 9786494.0, "reward": 0.07373046875, "reward_std": 6.209069728851318, "rewards/rm_reward_func/mean": 0.07373046875, "rewards/rm_reward_func/std": 12.3759765625, "step": 736 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 330.8125, "completions/mean_terminated_length": 324.9677429199219, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "epoch": 0.5896, "grad_norm": 5.071633815765381, "kl": 1.260986328125, "learning_rate": 1e-06, "loss": -0.0572, "num_tokens": 9799144.0, "reward": 9.56390380859375, "reward_std": 10.431290626525879, "rewards/rm_reward_func/mean": 9.56390380859375, "rewards/rm_reward_func/std": 19.839946746826172, "step": 737 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 380.25, "completions/mean_terminated_length": 328.6956481933594, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "epoch": 0.5904, "grad_norm": 10.128765106201172, "kl": 3.7099609375, "learning_rate": 1e-06, "loss": 0.1282, "num_tokens": 9816376.0, "reward": 7.971923828125, "reward_std": 5.655712127685547, "rewards/rm_reward_func/mean": 7.971923828125, "rewards/rm_reward_func/std": 19.648178100585938, "step": 738 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 320.96875, "completions/mean_terminated_length": 276.8846130371094, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "epoch": 0.5912, "grad_norm": 12.91031551361084, "kl": 1.35986328125, "learning_rate": 1e-06, "loss": 0.0801, "num_tokens": 9832231.0, "reward": 14.126220703125, "reward_std": 8.189908981323242, "rewards/rm_reward_func/mean": 14.126220703125, "rewards/rm_reward_func/std": 11.697257995605469, "step": 739 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 344.125, "completions/mean_terminated_length": 288.16668701171875, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "epoch": 0.592, "grad_norm": 6.711599826812744, "kl": 6.6611328125, "learning_rate": 1e-06, "loss": 0.4934, "num_tokens": 9846835.0, "reward": 3.853271484375, "reward_std": 11.102495193481445, "rewards/rm_reward_func/mean": 3.853271484375, "rewards/rm_reward_func/std": 13.937686920166016, "step": 740 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 369.5, "completions/mean_terminated_length": 294.8571472167969, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "epoch": 0.5928, "grad_norm": 9.946474075317383, "kl": 3.537353515625, "learning_rate": 1e-06, "loss": 0.2248, "num_tokens": 9863235.0, "reward": 5.8260498046875, "reward_std": 8.534289360046387, "rewards/rm_reward_func/mean": 5.8260498046875, "rewards/rm_reward_func/std": 14.377345085144043, "step": 741 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 261.8125, "completions/mean_terminated_length": 204.07693481445312, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "epoch": 0.5936, "grad_norm": 34.741207122802734, "kl": 16.79248046875, "learning_rate": 1e-06, "loss": 0.9002, "num_tokens": 9875661.0, "reward": -8.799308776855469, "reward_std": 6.605631351470947, "rewards/rm_reward_func/mean": -8.799308776855469, "rewards/rm_reward_func/std": 15.544058799743652, "step": 742 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 390.09375, "completions/mean_terminated_length": 316.95001220703125, "completions/min_length": 100.0, "completions/min_terminated_length": 100.0, "epoch": 0.5944, "grad_norm": 7.391372203826904, "kl": 0.4296875, "learning_rate": 1e-06, "loss": 0.0217, "num_tokens": 9891272.0, "reward": 8.581817626953125, "reward_std": 3.363267183303833, "rewards/rm_reward_func/mean": 8.581817626953125, "rewards/rm_reward_func/std": 12.903508186340332, "step": 743 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 262.15625, "completions/mean_terminated_length": 254.09677124023438, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "epoch": 0.5952, "grad_norm": 6.578904151916504, "kl": 5.13134765625, "learning_rate": 1e-06, "loss": 0.4582, "num_tokens": 9904589.0, "reward": 18.150390625, "reward_std": 6.425145626068115, "rewards/rm_reward_func/mean": 18.150390625, "rewards/rm_reward_func/std": 10.988823890686035, "step": 744 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 450.0, "completions/mean_length": 300.03125, "completions/mean_terminated_length": 251.11538696289062, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "epoch": 0.596, "grad_norm": 26.791982650756836, "kl": 5.4140625, "learning_rate": 1e-06, "loss": 0.1141, "num_tokens": 9918158.0, "reward": -5.0845947265625, "reward_std": 7.70248556137085, "rewards/rm_reward_func/mean": -5.0845947265625, "rewards/rm_reward_func/std": 11.328791618347168, "step": 745 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 461.0, "completions/mean_length": 314.125, "completions/mean_terminated_length": 268.4615478515625, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "epoch": 0.5968, "grad_norm": 10.60112476348877, "kl": 2.5537109375, "learning_rate": 1e-06, "loss": -0.0021, "num_tokens": 9930418.0, "reward": 3.98583984375, "reward_std": 7.434886932373047, "rewards/rm_reward_func/mean": 3.98583984375, "rewards/rm_reward_func/std": 15.446136474609375, "step": 746 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 408.5625, "completions/mean_terminated_length": 397.862060546875, "completions/min_length": 251.0, "completions/min_terminated_length": 251.0, "epoch": 0.5976, "grad_norm": 7.734127998352051, "kl": 1.279541015625, "learning_rate": 1e-06, "loss": 0.0852, "num_tokens": 9947396.0, "reward": 8.797119140625, "reward_std": 9.704774856567383, "rewards/rm_reward_func/mean": 8.797119140625, "rewards/rm_reward_func/std": 14.60503101348877, "step": 747 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 305.9375, "completions/mean_terminated_length": 299.2903137207031, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "epoch": 0.5984, "grad_norm": 30.35190200805664, "kl": 10.0810546875, "learning_rate": 1e-06, "loss": 0.2479, "num_tokens": 9959546.0, "reward": -7.094024658203125, "reward_std": 9.547418594360352, "rewards/rm_reward_func/mean": -7.094024658203125, "rewards/rm_reward_func/std": 12.000333786010742, "step": 748 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 285.0, "completions/mean_terminated_length": 277.6773986816406, "completions/min_length": 111.0, "completions/min_terminated_length": 111.0, "epoch": 0.5992, "grad_norm": 9.810269355773926, "kl": 0.603515625, "learning_rate": 1e-06, "loss": 0.057, "num_tokens": 9970658.0, "reward": 2.1044921875, "reward_std": 3.5760817527770996, "rewards/rm_reward_func/mean": 2.1044921875, "rewards/rm_reward_func/std": 6.474411487579346, "step": 749 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 356.4375, "completions/mean_terminated_length": 295.5652160644531, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "epoch": 0.6, "grad_norm": 14.325955390930176, "kl": 3.8134765625, "learning_rate": 1e-06, "loss": 0.0913, "num_tokens": 9984464.0, "reward": 4.713623046875, "reward_std": 8.757728576660156, "rewards/rm_reward_func/mean": 4.713623046875, "rewards/rm_reward_func/std": 16.11822509765625, "step": 750 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 297.5625, "completions/mean_terminated_length": 275.3793029785156, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "epoch": 0.6008, "grad_norm": 44.704063415527344, "kl": 3.189453125, "learning_rate": 1e-06, "loss": 0.2095, "num_tokens": 9997474.0, "reward": 10.037109375, "reward_std": 13.35189151763916, "rewards/rm_reward_func/mean": 10.037109375, "rewards/rm_reward_func/std": 21.836666107177734, "step": 751 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 264.59375, "completions/mean_terminated_length": 256.6128845214844, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "epoch": 0.6016, "grad_norm": 7.517630100250244, "kl": 4.232177734375, "learning_rate": 1e-06, "loss": 0.215, "num_tokens": 10008461.0, "reward": 0.819580078125, "reward_std": 10.651954650878906, "rewards/rm_reward_func/mean": 0.819580078125, "rewards/rm_reward_func/std": 15.250157356262207, "step": 752 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 293.25, "completions/mean_terminated_length": 278.66668701171875, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "epoch": 0.6024, "grad_norm": 10.266236305236816, "kl": 4.54296875, "learning_rate": 1e-06, "loss": 0.0359, "num_tokens": 10020021.0, "reward": -3.8173828125, "reward_std": 9.468101501464844, "rewards/rm_reward_func/mean": -3.8173828125, "rewards/rm_reward_func/std": 13.068405151367188, "step": 753 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 477.0, "completions/mean_length": 192.4375, "completions/mean_terminated_length": 146.7857208251953, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "epoch": 0.6032, "grad_norm": 9.960123062133789, "kl": 5.95947265625, "learning_rate": 1e-06, "loss": 0.0333, "num_tokens": 10030379.0, "reward": -3.7373046875, "reward_std": 10.038860321044922, "rewards/rm_reward_func/mean": -3.7373046875, "rewards/rm_reward_func/std": 15.353011131286621, "step": 754 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 463.0, "completions/mean_length": 317.65625, "completions/mean_terminated_length": 241.60870361328125, "completions/min_length": 111.0, "completions/min_terminated_length": 111.0, "epoch": 0.604, "grad_norm": 6.909921169281006, "kl": 1.919921875, "learning_rate": 1e-06, "loss": 0.0276, "num_tokens": 10043216.0, "reward": 1.073760986328125, "reward_std": 6.802268028259277, "rewards/rm_reward_func/mean": 1.073760986328125, "rewards/rm_reward_func/std": 10.724485397338867, "step": 755 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 486.0, "completions/mean_length": 322.78125, "completions/mean_terminated_length": 310.16668701171875, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "epoch": 0.6048, "grad_norm": 4.838641166687012, "kl": 1.0595703125, "learning_rate": 1e-06, "loss": -0.0381, "num_tokens": 10057593.0, "reward": 16.49951171875, "reward_std": 8.707597732543945, "rewards/rm_reward_func/mean": 16.49951171875, "rewards/rm_reward_func/std": 13.02664566040039, "step": 756 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 232.6875, "completions/mean_terminated_length": 223.6774139404297, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "epoch": 0.6056, "grad_norm": 6.288021087646484, "kl": 2.422119140625, "learning_rate": 1e-06, "loss": 0.0256, "num_tokens": 10068959.0, "reward": 7.3639373779296875, "reward_std": 7.31492805480957, "rewards/rm_reward_func/mean": 7.3639373779296875, "rewards/rm_reward_func/std": 11.444160461425781, "step": 757 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 461.0, "completions/mean_length": 257.375, "completions/mean_terminated_length": 240.40000915527344, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "epoch": 0.6064, "grad_norm": 12.485620498657227, "kl": 1.30322265625, "learning_rate": 1e-06, "loss": -0.256, "num_tokens": 10079523.0, "reward": 8.375, "reward_std": 11.750480651855469, "rewards/rm_reward_func/mean": 8.375, "rewards/rm_reward_func/std": 15.094318389892578, "step": 758 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 359.3125, "completions/mean_terminated_length": 308.41668701171875, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "epoch": 0.6072, "grad_norm": 67.01472473144531, "kl": 0.41552734375, "learning_rate": 1e-06, "loss": -0.0451, "num_tokens": 10096813.0, "reward": 8.5546875, "reward_std": 4.309829235076904, "rewards/rm_reward_func/mean": 8.5546875, "rewards/rm_reward_func/std": 15.106447219848633, "step": 759 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 387.09375, "completions/mean_terminated_length": 352.1199951171875, "completions/min_length": 169.0, "completions/min_terminated_length": 169.0, "epoch": 0.608, "grad_norm": 7.828210830688477, "kl": 1.158203125, "learning_rate": 1e-06, "loss": 0.0712, "num_tokens": 10112304.0, "reward": 2.0579833984375, "reward_std": 6.627885341644287, "rewards/rm_reward_func/mean": 2.0579833984375, "rewards/rm_reward_func/std": 9.508418083190918, "step": 760 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 299.96875, "completions/mean_terminated_length": 240.59999084472656, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "epoch": 0.6088, "grad_norm": 32.08409118652344, "kl": 9.375, "learning_rate": 1e-06, "loss": 0.0282, "num_tokens": 10124279.0, "reward": -8.2109375, "reward_std": 14.024812698364258, "rewards/rm_reward_func/mean": -8.2109375, "rewards/rm_reward_func/std": 16.273033142089844, "step": 761 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 475.0, "completions/mean_length": 267.03125, "completions/mean_terminated_length": 185.375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "epoch": 0.6096, "grad_norm": 9.463338851928711, "kl": 0.489013671875, "learning_rate": 1e-06, "loss": 0.0646, "num_tokens": 10135368.0, "reward": 4.3076171875, "reward_std": 3.226278781890869, "rewards/rm_reward_func/mean": 4.3076171875, "rewards/rm_reward_func/std": 11.224862098693848, "step": 762 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 210.40625, "completions/mean_terminated_length": 210.40625, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "epoch": 0.6104, "grad_norm": 15.222582817077637, "kl": 2.01904296875, "learning_rate": 1e-06, "loss": 0.1217, "num_tokens": 10144741.0, "reward": -5.0877685546875, "reward_std": 11.080349922180176, "rewards/rm_reward_func/mean": -5.0877685546875, "rewards/rm_reward_func/std": 13.31748104095459, "step": 763 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 414.3125, "completions/mean_terminated_length": 386.9599914550781, "completions/min_length": 259.0, "completions/min_terminated_length": 259.0, "epoch": 0.6112, "grad_norm": 13.568563461303711, "kl": 3.62890625, "learning_rate": 1e-06, "loss": 0.0975, "num_tokens": 10160167.0, "reward": 6.3486328125, "reward_std": 12.485017776489258, "rewards/rm_reward_func/mean": 6.3486328125, "rewards/rm_reward_func/std": 13.953908920288086, "step": 764 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 378.0, "completions/max_terminated_length": 378.0, "completions/mean_length": 162.1875, "completions/mean_terminated_length": 162.1875, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "epoch": 0.612, "grad_norm": 19.0142765045166, "kl": 2.60107421875, "learning_rate": 1e-06, "loss": 0.0385, "num_tokens": 10170917.0, "reward": -2.573974609375, "reward_std": 3.245800495147705, "rewards/rm_reward_func/mean": -2.573974609375, "rewards/rm_reward_func/std": 13.304573059082031, "step": 765 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.53125, "completions/max_length": 512.0, "completions/max_terminated_length": 491.0, "completions/mean_length": 436.625, "completions/mean_terminated_length": 351.20001220703125, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "epoch": 0.6128, "grad_norm": 24.08083724975586, "kl": 2.4443359375, "learning_rate": 1e-06, "loss": 0.0583, "num_tokens": 10190281.0, "reward": 2.33367919921875, "reward_std": 6.308701992034912, "rewards/rm_reward_func/mean": 2.33367919921875, "rewards/rm_reward_func/std": 15.869560241699219, "step": 766 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 345.46875, "completions/mean_terminated_length": 280.3043518066406, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "epoch": 0.6136, "grad_norm": 9.395527839660645, "kl": 2.9775390625, "learning_rate": 1e-06, "loss": 0.0412, "num_tokens": 10205904.0, "reward": -1.7017822265625, "reward_std": 6.9784321784973145, "rewards/rm_reward_func/mean": -1.7017822265625, "rewards/rm_reward_func/std": 11.290467262268066, "step": 767 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 479.0, "completions/mean_length": 262.09375, "completions/mean_terminated_length": 131.1904754638672, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "epoch": 0.6144, "grad_norm": 6.257786750793457, "kl": 1.263671875, "learning_rate": 1e-06, "loss": -0.0265, "num_tokens": 10219307.0, "reward": 0.228759765625, "reward_std": 3.712106466293335, "rewards/rm_reward_func/mean": 0.228759765625, "rewards/rm_reward_func/std": 7.532781600952148, "step": 768 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 469.0, "completions/mean_length": 219.875, "completions/mean_terminated_length": 200.40000915527344, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "epoch": 0.6152, "grad_norm": 12.303415298461914, "kl": 1.0126953125, "learning_rate": 1e-06, "loss": -0.1391, "num_tokens": 10229703.0, "reward": 1.8739013671875, "reward_std": 9.617498397827148, "rewards/rm_reward_func/mean": 1.8739013671875, "rewards/rm_reward_func/std": 12.91628360748291, "step": 769 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 399.0625, "completions/mean_terminated_length": 367.44000244140625, "completions/min_length": 95.0, "completions/min_terminated_length": 95.0, "epoch": 0.616, "grad_norm": 7.800823211669922, "kl": 2.70703125, "learning_rate": 1e-06, "loss": 0.0288, "num_tokens": 10244537.0, "reward": 12.45703125, "reward_std": 13.960775375366211, "rewards/rm_reward_func/mean": 12.45703125, "rewards/rm_reward_func/std": 23.239635467529297, "step": 770 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 492.0, "completions/mean_length": 414.78125, "completions/mean_terminated_length": 376.7391357421875, "completions/min_length": 184.0, "completions/min_terminated_length": 184.0, "epoch": 0.6168, "grad_norm": 23.5693302154541, "kl": 1.23095703125, "learning_rate": 1e-06, "loss": 0.0611, "num_tokens": 10260738.0, "reward": 7.40087890625, "reward_std": 7.460350036621094, "rewards/rm_reward_func/mean": 7.40087890625, "rewards/rm_reward_func/std": 15.638055801391602, "step": 771 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 442.0, "completions/mean_length": 293.3125, "completions/mean_terminated_length": 262.0714416503906, "completions/min_length": 82.0, "completions/min_terminated_length": 82.0, "epoch": 0.6176, "grad_norm": 24.51900291442871, "kl": 1.25, "learning_rate": 1e-06, "loss": -0.0135, "num_tokens": 10275532.0, "reward": 7.166015625, "reward_std": 10.221585273742676, "rewards/rm_reward_func/mean": 7.166015625, "rewards/rm_reward_func/std": 18.162363052368164, "step": 772 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 448.0, "completions/mean_length": 351.6875, "completions/mean_terminated_length": 341.0000305175781, "completions/min_length": 146.0, "completions/min_terminated_length": 146.0, "epoch": 0.6184, "grad_norm": 11.453636169433594, "kl": 0.533447265625, "learning_rate": 1e-06, "loss": -0.0055, "num_tokens": 10288858.0, "reward": 10.736896514892578, "reward_std": 6.508322715759277, "rewards/rm_reward_func/mean": 10.736896514892578, "rewards/rm_reward_func/std": 15.74055290222168, "step": 773 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 187.375, "completions/mean_terminated_length": 165.73333740234375, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "epoch": 0.6192, "grad_norm": 11.30697250366211, "kl": 2.5537109375, "learning_rate": 1e-06, "loss": 0.0897, "num_tokens": 10299606.0, "reward": -3.98828125, "reward_std": 3.1018943786621094, "rewards/rm_reward_func/mean": -3.98828125, "rewards/rm_reward_func/std": 13.498686790466309, "step": 774 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 507.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 319.6875, "completions/mean_terminated_length": 319.6875, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "epoch": 0.62, "grad_norm": 13.508712768554688, "kl": 5.896484375, "learning_rate": 1e-06, "loss": 0.2158, "num_tokens": 10312068.0, "reward": -2.490478515625, "reward_std": 13.193625450134277, "rewards/rm_reward_func/mean": -2.490478515625, "rewards/rm_reward_func/std": 20.5009765625, "step": 775 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 284.3125, "completions/mean_terminated_length": 220.55999755859375, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "epoch": 0.6208, "grad_norm": 30.201000213623047, "kl": 3.12548828125, "learning_rate": 1e-06, "loss": 0.0509, "num_tokens": 10324806.0, "reward": 6.44482421875, "reward_std": 9.712287902832031, "rewards/rm_reward_func/mean": 6.44482421875, "rewards/rm_reward_func/std": 14.30795669555664, "step": 776 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 299.1875, "completions/mean_terminated_length": 292.32257080078125, "completions/min_length": 40.0, "completions/min_terminated_length": 40.0, "epoch": 0.6216, "grad_norm": 33.898277282714844, "kl": 3.952392578125, "learning_rate": 1e-06, "loss": 0.0502, "num_tokens": 10337852.0, "reward": 0.22265625, "reward_std": 9.971935272216797, "rewards/rm_reward_func/mean": 0.22265625, "rewards/rm_reward_func/std": 19.976945877075195, "step": 777 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 470.0, "completions/mean_length": 257.09375, "completions/mean_terminated_length": 248.87095642089844, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "epoch": 0.6224, "grad_norm": 7.3413543701171875, "kl": 1.251708984375, "learning_rate": 1e-06, "loss": 0.005, "num_tokens": 10350439.0, "reward": 5.71923828125, "reward_std": 5.034487724304199, "rewards/rm_reward_func/mean": 5.71923828125, "rewards/rm_reward_func/std": 12.428587913513184, "step": 778 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 206.75, "completions/mean_terminated_length": 196.90321350097656, "completions/min_length": 20.0, "completions/min_terminated_length": 20.0, "epoch": 0.6232, "grad_norm": 15.728623390197754, "kl": 5.38427734375, "learning_rate": 1e-06, "loss": 0.185, "num_tokens": 10360047.0, "reward": -2.06689453125, "reward_std": 5.258633613586426, "rewards/rm_reward_func/mean": -2.06689453125, "rewards/rm_reward_func/std": 14.16489315032959, "step": 779 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 181.03125, "completions/mean_terminated_length": 170.35482788085938, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "epoch": 0.624, "grad_norm": 9.44076919555664, "kl": 0.400390625, "learning_rate": 1e-06, "loss": 0.0051, "num_tokens": 10369952.0, "reward": 6.325927734375, "reward_std": 2.282184600830078, "rewards/rm_reward_func/mean": 6.325927734375, "rewards/rm_reward_func/std": 5.712595462799072, "step": 780 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 485.0, "completions/mean_length": 298.90625, "completions/mean_terminated_length": 292.0322570800781, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "epoch": 0.6248, "grad_norm": 7.031983375549316, "kl": 3.04150390625, "learning_rate": 1e-06, "loss": -0.0902, "num_tokens": 10384941.0, "reward": 12.3427734375, "reward_std": 17.103315353393555, "rewards/rm_reward_func/mean": 12.3427734375, "rewards/rm_reward_func/std": 21.697189331054688, "step": 781 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 345.53125, "completions/mean_terminated_length": 334.433349609375, "completions/min_length": 104.0, "completions/min_terminated_length": 104.0, "epoch": 0.6256, "grad_norm": 12.419363021850586, "kl": 1.218994140625, "learning_rate": 1e-06, "loss": -0.085, "num_tokens": 10403062.0, "reward": 9.69482421875, "reward_std": 10.32385540008545, "rewards/rm_reward_func/mean": 9.69482421875, "rewards/rm_reward_func/std": 19.279569625854492, "step": 782 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 271.0625, "completions/mean_terminated_length": 236.6428680419922, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "epoch": 0.6264, "grad_norm": 13.157743453979492, "kl": 1.40673828125, "learning_rate": 1e-06, "loss": 0.0223, "num_tokens": 10414368.0, "reward": 5.6474609375, "reward_std": 5.8091206550598145, "rewards/rm_reward_func/mean": 5.6474609375, "rewards/rm_reward_func/std": 13.779108047485352, "step": 783 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 284.125, "completions/mean_terminated_length": 268.933349609375, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "epoch": 0.6272, "grad_norm": 4.267988681793213, "kl": 0.6849365234375, "learning_rate": 1e-06, "loss": 0.0464, "num_tokens": 10426644.0, "reward": 8.6116943359375, "reward_std": 8.114376068115234, "rewards/rm_reward_func/mean": 8.6116943359375, "rewards/rm_reward_func/std": 13.790817260742188, "step": 784 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 310.46875, "completions/mean_terminated_length": 231.60870361328125, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "epoch": 0.628, "grad_norm": 26.008434295654297, "kl": 1.810791015625, "learning_rate": 1e-06, "loss": 0.0316, "num_tokens": 10443571.0, "reward": 7.016845703125, "reward_std": 8.47242259979248, "rewards/rm_reward_func/mean": 7.016845703125, "rewards/rm_reward_func/std": 13.28702449798584, "step": 785 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 244.15625, "completions/mean_terminated_length": 216.44827270507812, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "epoch": 0.6288, "grad_norm": 18.478044509887695, "kl": 6.23046875, "learning_rate": 1e-06, "loss": 0.1698, "num_tokens": 10453696.0, "reward": 2.2125244140625, "reward_std": 8.634559631347656, "rewards/rm_reward_func/mean": 2.2125244140625, "rewards/rm_reward_func/std": 14.19184684753418, "step": 786 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 463.0, "completions/mean_length": 280.875, "completions/mean_terminated_length": 238.07408142089844, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "epoch": 0.6296, "grad_norm": 8.919168472290039, "kl": 0.99560546875, "learning_rate": 1e-06, "loss": -0.0556, "num_tokens": 10467428.0, "reward": -6.05810546875, "reward_std": 5.291695594787598, "rewards/rm_reward_func/mean": -6.05810546875, "rewards/rm_reward_func/std": 14.077133178710938, "step": 787 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 350.96875, "completions/mean_terminated_length": 297.29168701171875, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "epoch": 0.6304, "grad_norm": 10.0316743850708, "kl": 4.890625, "learning_rate": 1e-06, "loss": -0.1151, "num_tokens": 10480643.0, "reward": -2.29998779296875, "reward_std": 18.708683013916016, "rewards/rm_reward_func/mean": -2.29998779296875, "rewards/rm_reward_func/std": 21.16217041015625, "step": 788 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 479.0, "completions/mean_length": 298.5, "completions/mean_terminated_length": 284.2666931152344, "completions/min_length": 16.0, "completions/min_terminated_length": 16.0, "epoch": 0.6312, "grad_norm": 21.06656837463379, "kl": 6.18212890625, "learning_rate": 1e-06, "loss": 0.3144, "num_tokens": 10492979.0, "reward": -6.215576171875, "reward_std": 7.852952003479004, "rewards/rm_reward_func/mean": -6.215576171875, "rewards/rm_reward_func/std": 14.924491882324219, "step": 789 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 495.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 184.15625, "completions/mean_terminated_length": 184.15625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "epoch": 0.632, "grad_norm": 6.202207565307617, "kl": 1.20654296875, "learning_rate": 1e-06, "loss": 0.0211, "num_tokens": 10503632.0, "reward": 7.91845703125, "reward_std": 6.895031452178955, "rewards/rm_reward_func/mean": 7.91845703125, "rewards/rm_reward_func/std": 13.781877517700195, "step": 790 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 200.8125, "completions/mean_terminated_length": 168.6206817626953, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "epoch": 0.6328, "grad_norm": 19.258525848388672, "kl": 2.4609375, "learning_rate": 1e-06, "loss": -0.188, "num_tokens": 10512882.0, "reward": 0.01904296875, "reward_std": 13.03593635559082, "rewards/rm_reward_func/mean": 0.01904296875, "rewards/rm_reward_func/std": 16.95264434814453, "step": 791 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 376.71875, "completions/mean_terminated_length": 331.625, "completions/min_length": 130.0, "completions/min_terminated_length": 130.0, "epoch": 0.6336, "grad_norm": 15.64734172821045, "kl": 5.783203125, "learning_rate": 1e-06, "loss": 0.1589, "num_tokens": 10527289.0, "reward": 0.9697265625, "reward_std": 10.092090606689453, "rewards/rm_reward_func/mean": 0.9697265625, "rewards/rm_reward_func/std": 18.045101165771484, "step": 792 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 442.59375, "completions/mean_terminated_length": 419.4583435058594, "completions/min_length": 307.0, "completions/min_terminated_length": 307.0, "epoch": 0.6344, "grad_norm": 5.187870025634766, "kl": 2.06005859375, "learning_rate": 1e-06, "loss": 0.0698, "num_tokens": 10543972.0, "reward": 14.57086181640625, "reward_std": 9.486396789550781, "rewards/rm_reward_func/mean": 14.57086181640625, "rewards/rm_reward_func/std": 18.273242950439453, "step": 793 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 387.5625, "completions/mean_terminated_length": 364.5185241699219, "completions/min_length": 173.0, "completions/min_terminated_length": 173.0, "epoch": 0.6352, "grad_norm": 14.29277229309082, "kl": 4.53125, "learning_rate": 1e-06, "loss": 0.1114, "num_tokens": 10560582.0, "reward": -0.26617431640625, "reward_std": 8.553365707397461, "rewards/rm_reward_func/mean": -0.26617431640625, "rewards/rm_reward_func/std": 10.436263084411621, "step": 794 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 236.59375, "completions/mean_terminated_length": 227.7096710205078, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "epoch": 0.636, "grad_norm": 7.5032501220703125, "kl": 1.154296875, "learning_rate": 1e-06, "loss": 0.0065, "num_tokens": 10571985.0, "reward": 7.23681640625, "reward_std": 8.498395919799805, "rewards/rm_reward_func/mean": 7.23681640625, "rewards/rm_reward_func/std": 13.58292007446289, "step": 795 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 448.0, "completions/mean_length": 239.15625, "completions/mean_terminated_length": 162.75999450683594, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "epoch": 0.6368, "grad_norm": 21.288700103759766, "kl": 6.2919921875, "learning_rate": 1e-06, "loss": 0.1455, "num_tokens": 10582014.0, "reward": -8.60565185546875, "reward_std": 5.36991548538208, "rewards/rm_reward_func/mean": -8.60565185546875, "rewards/rm_reward_func/std": 15.003857612609863, "step": 796 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 247.03125, "completions/mean_terminated_length": 209.17857360839844, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "epoch": 0.6376, "grad_norm": 7.441081523895264, "kl": 2.571533203125, "learning_rate": 1e-06, "loss": 0.0626, "num_tokens": 10592679.0, "reward": 2.77825927734375, "reward_std": 2.551882028579712, "rewards/rm_reward_func/mean": 2.77825927734375, "rewards/rm_reward_func/std": 8.97035026550293, "step": 797 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 176.28125, "completions/mean_terminated_length": 165.4516143798828, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "epoch": 0.6384, "grad_norm": 6.360789775848389, "kl": 0.4638671875, "learning_rate": 1e-06, "loss": 0.0344, "num_tokens": 10604672.0, "reward": 15.427978515625, "reward_std": 2.812357187271118, "rewards/rm_reward_func/mean": 15.427978515625, "rewards/rm_reward_func/std": 14.900362014770508, "step": 798 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 476.0, "completions/mean_length": 338.125, "completions/mean_terminated_length": 259.0909118652344, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "epoch": 0.6392, "grad_norm": 6.797297954559326, "kl": 0.995361328125, "learning_rate": 1e-06, "loss": 0.0529, "num_tokens": 10621628.0, "reward": 0.404541015625, "reward_std": 7.615126609802246, "rewards/rm_reward_func/mean": 0.404541015625, "rewards/rm_reward_func/std": 12.74864673614502, "step": 799 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 421.46875, "completions/mean_terminated_length": 374.0476379394531, "completions/min_length": 155.0, "completions/min_terminated_length": 155.0, "epoch": 0.64, "grad_norm": 9.847874641418457, "kl": 2.7900390625, "learning_rate": 1e-06, "loss": 0.0953, "num_tokens": 10638051.0, "reward": 3.163330078125, "reward_std": 10.612391471862793, "rewards/rm_reward_func/mean": 3.163330078125, "rewards/rm_reward_func/std": 15.437744140625, "step": 800 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 455.0, "completions/mean_length": 216.84375, "completions/mean_terminated_length": 148.73077392578125, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "epoch": 0.6408, "grad_norm": 10.988313674926758, "kl": 3.015625, "learning_rate": 1e-06, "loss": 0.1434, "num_tokens": 10647630.0, "reward": -3.7789306640625, "reward_std": 6.626611232757568, "rewards/rm_reward_func/mean": -3.7789306640625, "rewards/rm_reward_func/std": 10.867609977722168, "step": 801 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 401.0625, "completions/mean_terminated_length": 364.0833435058594, "completions/min_length": 97.0, "completions/min_terminated_length": 97.0, "epoch": 0.6416, "grad_norm": 6.588342666625977, "kl": 1.3544921875, "learning_rate": 1e-06, "loss": -0.0717, "num_tokens": 10665272.0, "reward": 3.876220703125, "reward_std": 8.485553741455078, "rewards/rm_reward_func/mean": 3.876220703125, "rewards/rm_reward_func/std": 11.902339935302734, "step": 802 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 264.9375, "completions/mean_terminated_length": 256.9677429199219, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "epoch": 0.6424, "grad_norm": 16.689680099487305, "kl": 4.962890625, "learning_rate": 1e-06, "loss": 0.2397, "num_tokens": 10676294.0, "reward": -3.4169921875, "reward_std": 6.7452592849731445, "rewards/rm_reward_func/mean": -3.4169921875, "rewards/rm_reward_func/std": 19.948022842407227, "step": 803 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 485.0, "completions/max_terminated_length": 485.0, "completions/mean_length": 270.46875, "completions/mean_terminated_length": 270.46875, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "epoch": 0.6432, "grad_norm": 6.953851222991943, "kl": 1.05712890625, "learning_rate": 1e-06, "loss": -0.0594, "num_tokens": 10690029.0, "reward": -0.96844482421875, "reward_std": 4.365692138671875, "rewards/rm_reward_func/mean": -0.96844482421875, "rewards/rm_reward_func/std": 6.488677024841309, "step": 804 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 277.65625, "completions/mean_terminated_length": 270.0967712402344, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "epoch": 0.644, "grad_norm": 6.296433448791504, "kl": 1.06787109375, "learning_rate": 1e-06, "loss": -0.0623, "num_tokens": 10701498.0, "reward": 1.8621826171875, "reward_std": 4.512002468109131, "rewards/rm_reward_func/mean": 1.8621826171875, "rewards/rm_reward_func/std": 13.205521583557129, "step": 805 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 211.0625, "completions/mean_terminated_length": 201.35482788085938, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "epoch": 0.6448, "grad_norm": 6.924914836883545, "kl": 2.736328125, "learning_rate": 1e-06, "loss": 0.0839, "num_tokens": 10712828.0, "reward": 4.1123046875, "reward_std": 2.182391405105591, "rewards/rm_reward_func/mean": 4.1123046875, "rewards/rm_reward_func/std": 6.567470073699951, "step": 806 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 256.65625, "completions/mean_terminated_length": 209.37037658691406, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "epoch": 0.6456, "grad_norm": 18.669151306152344, "kl": 6.09765625, "learning_rate": 1e-06, "loss": 0.2974, "num_tokens": 10724233.0, "reward": -15.416748046875, "reward_std": 7.353342056274414, "rewards/rm_reward_func/mean": -15.416748046875, "rewards/rm_reward_func/std": 11.860098838806152, "step": 807 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 482.0, "completions/mean_length": 396.25, "completions/mean_terminated_length": 326.8000183105469, "completions/min_length": 140.0, "completions/min_terminated_length": 140.0, "epoch": 0.6464, "grad_norm": 6.7639875411987305, "kl": 4.123046875, "learning_rate": 1e-06, "loss": 0.1232, "num_tokens": 10739529.0, "reward": -0.310546875, "reward_std": 9.28040885925293, "rewards/rm_reward_func/mean": -0.310546875, "rewards/rm_reward_func/std": 11.789945602416992, "step": 808 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 353.40625, "completions/mean_terminated_length": 324.03704833984375, "completions/min_length": 136.0, "completions/min_terminated_length": 136.0, "epoch": 0.6472, "grad_norm": 11.955965042114258, "kl": 5.74609375, "learning_rate": 1e-06, "loss": 0.123, "num_tokens": 10756438.0, "reward": 1.25439453125, "reward_std": 12.41511058807373, "rewards/rm_reward_func/mean": 1.25439453125, "rewards/rm_reward_func/std": 13.135236740112305, "step": 809 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 475.0, "completions/mean_length": 262.03125, "completions/mean_terminated_length": 236.1724090576172, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "epoch": 0.648, "grad_norm": 6.4916911125183105, "kl": 4.392578125, "learning_rate": 1e-06, "loss": 0.1702, "num_tokens": 10767359.0, "reward": -5.23974609375, "reward_std": 5.122437477111816, "rewards/rm_reward_func/mean": -5.23974609375, "rewards/rm_reward_func/std": 10.06690502166748, "step": 810 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 358.96875, "completions/mean_terminated_length": 330.629638671875, "completions/min_length": 110.0, "completions/min_terminated_length": 110.0, "epoch": 0.6488, "grad_norm": 38.71643829345703, "kl": 10.5546875, "learning_rate": 1e-06, "loss": 0.3489, "num_tokens": 10780798.0, "reward": -9.0185546875, "reward_std": 5.921327590942383, "rewards/rm_reward_func/mean": -9.0185546875, "rewards/rm_reward_func/std": 9.659137725830078, "step": 811 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 368.40625, "completions/mean_terminated_length": 358.8333435058594, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "epoch": 0.6496, "grad_norm": 8.951486587524414, "kl": 3.515625, "learning_rate": 1e-06, "loss": -0.0127, "num_tokens": 10795059.0, "reward": -1.08935546875, "reward_std": 11.639703750610352, "rewards/rm_reward_func/mean": -1.08935546875, "rewards/rm_reward_func/std": 14.727822303771973, "step": 812 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 480.0, "completions/mean_length": 289.21875, "completions/mean_terminated_length": 237.80770874023438, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "epoch": 0.6504, "grad_norm": 21.51280403137207, "kl": 4.89453125, "learning_rate": 1e-06, "loss": 0.0933, "num_tokens": 10806426.0, "reward": -5.33056640625, "reward_std": 8.251094818115234, "rewards/rm_reward_func/mean": -5.33056640625, "rewards/rm_reward_func/std": 14.58669376373291, "step": 813 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 485.0, "completions/mean_length": 325.90625, "completions/mean_terminated_length": 273.79998779296875, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "epoch": 0.6512, "grad_norm": 14.707463264465332, "kl": 6.0205078125, "learning_rate": 1e-06, "loss": 0.1921, "num_tokens": 10820431.0, "reward": -2.914306640625, "reward_std": 8.024243354797363, "rewards/rm_reward_func/mean": -2.914306640625, "rewards/rm_reward_func/std": 14.138604164123535, "step": 814 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 264.0625, "completions/mean_terminated_length": 256.06451416015625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "epoch": 0.652, "grad_norm": 9.099730491638184, "kl": 2.73681640625, "learning_rate": 1e-06, "loss": 0.0878, "num_tokens": 10831561.0, "reward": -2.70703125, "reward_std": 6.87619686126709, "rewards/rm_reward_func/mean": -2.70703125, "rewards/rm_reward_func/std": 13.68471908569336, "step": 815 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 276.15625, "completions/mean_terminated_length": 260.433349609375, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "epoch": 0.6528, "grad_norm": 9.196378707885742, "kl": 4.783203125, "learning_rate": 1e-06, "loss": 0.0501, "num_tokens": 10843046.0, "reward": -3.22216796875, "reward_std": 6.319431781768799, "rewards/rm_reward_func/mean": -3.22216796875, "rewards/rm_reward_func/std": 8.792978286743164, "step": 816 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 403.0, "completions/max_terminated_length": 403.0, "completions/mean_length": 156.75, "completions/mean_terminated_length": 156.75, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "epoch": 0.6536, "grad_norm": 7.186251163482666, "kl": 5.7412109375, "learning_rate": 1e-06, "loss": 0.248, "num_tokens": 10851550.0, "reward": 1.57763671875, "reward_std": 8.453566551208496, "rewards/rm_reward_func/mean": 1.57763671875, "rewards/rm_reward_func/std": 19.387672424316406, "step": 817 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 292.71875, "completions/mean_terminated_length": 270.03448486328125, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "epoch": 0.6544, "grad_norm": 9.593155860900879, "kl": 1.69677734375, "learning_rate": 1e-06, "loss": 0.0271, "num_tokens": 10865029.0, "reward": 5.674560546875, "reward_std": 6.335400581359863, "rewards/rm_reward_func/mean": 5.674560546875, "rewards/rm_reward_func/std": 13.289620399475098, "step": 818 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 442.0, "completions/mean_length": 311.0, "completions/mean_terminated_length": 244.0, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "epoch": 0.6552, "grad_norm": 9.963946342468262, "kl": 1.982421875, "learning_rate": 1e-06, "loss": 0.0487, "num_tokens": 10877253.0, "reward": -0.123199462890625, "reward_std": 6.386770248413086, "rewards/rm_reward_func/mean": -0.123199462890625, "rewards/rm_reward_func/std": 10.461974143981934, "step": 819 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 426.0, "completions/mean_length": 358.9375, "completions/mean_terminated_length": 307.91668701171875, "completions/min_length": 156.0, "completions/min_terminated_length": 156.0, "epoch": 0.656, "grad_norm": 6.363739013671875, "kl": 2.058837890625, "learning_rate": 1e-06, "loss": 0.1152, "num_tokens": 10893083.0, "reward": -0.5570068359375, "reward_std": 6.5521240234375, "rewards/rm_reward_func/mean": -0.5570068359375, "rewards/rm_reward_func/std": 18.305418014526367, "step": 820 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 319.75, "completions/mean_terminated_length": 299.862060546875, "completions/min_length": 151.0, "completions/min_terminated_length": 151.0, "epoch": 0.6568, "grad_norm": 8.368305206298828, "kl": 3.38720703125, "learning_rate": 1e-06, "loss": -0.0173, "num_tokens": 10906779.0, "reward": 4.049560546875, "reward_std": 14.061559677124023, "rewards/rm_reward_func/mean": 4.049560546875, "rewards/rm_reward_func/std": 21.243824005126953, "step": 821 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 486.0, "completions/mean_length": 324.6875, "completions/mean_terminated_length": 290.0, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "epoch": 0.6576, "grad_norm": 8.28355598449707, "kl": 2.156005859375, "learning_rate": 1e-06, "loss": -0.1195, "num_tokens": 10919833.0, "reward": -12.810791015625, "reward_std": 5.67118501663208, "rewards/rm_reward_func/mean": -12.810791015625, "rewards/rm_reward_func/std": 7.540193557739258, "step": 822 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 424.0, "completions/max_terminated_length": 424.0, "completions/mean_length": 167.9375, "completions/mean_terminated_length": 167.9375, "completions/min_length": 19.0, "completions/min_terminated_length": 19.0, "epoch": 0.6584, "grad_norm": 7.692931175231934, "kl": 4.6162109375, "learning_rate": 1e-06, "loss": 0.2633, "num_tokens": 10932023.0, "reward": -4.395263671875, "reward_std": 5.285364151000977, "rewards/rm_reward_func/mean": -4.395263671875, "rewards/rm_reward_func/std": 9.415786743164062, "step": 823 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 262.84375, "completions/mean_terminated_length": 193.0800018310547, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "epoch": 0.6592, "grad_norm": 11.74027156829834, "kl": 0.53271484375, "learning_rate": 1e-06, "loss": -0.1976, "num_tokens": 10943026.0, "reward": 4.1466064453125, "reward_std": 7.172249794006348, "rewards/rm_reward_func/mean": 4.1466064453125, "rewards/rm_reward_func/std": 11.245109558105469, "step": 824 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 314.4375, "completions/mean_terminated_length": 237.13043212890625, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "epoch": 0.66, "grad_norm": 4.412758827209473, "kl": 0.615966796875, "learning_rate": 1e-06, "loss": 0.0277, "num_tokens": 10957904.0, "reward": -2.131103515625, "reward_std": 4.323854446411133, "rewards/rm_reward_func/mean": -2.131103515625, "rewards/rm_reward_func/std": 8.868707656860352, "step": 825 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 462.0, "completions/mean_length": 349.3125, "completions/mean_terminated_length": 319.1851806640625, "completions/min_length": 119.0, "completions/min_terminated_length": 119.0, "epoch": 0.6608, "grad_norm": 8.348759651184082, "kl": 1.59619140625, "learning_rate": 1e-06, "loss": 0.065, "num_tokens": 10971842.0, "reward": 0.430908203125, "reward_std": 5.89361047744751, "rewards/rm_reward_func/mean": 0.430908203125, "rewards/rm_reward_func/std": 11.598136901855469, "step": 826 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 478.0, "completions/max_terminated_length": 478.0, "completions/mean_length": 261.21875, "completions/mean_terminated_length": 261.21875, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "epoch": 0.6616, "grad_norm": 10.604769706726074, "kl": 1.353515625, "learning_rate": 1e-06, "loss": -0.1187, "num_tokens": 10982777.0, "reward": 1.006195068359375, "reward_std": 13.28079891204834, "rewards/rm_reward_func/mean": 1.006195068359375, "rewards/rm_reward_func/std": 14.09341049194336, "step": 827 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 333.65625, "completions/mean_terminated_length": 321.7666931152344, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "epoch": 0.6624, "grad_norm": 6.49513053894043, "kl": 2.354248046875, "learning_rate": 1e-06, "loss": 0.1536, "num_tokens": 10996190.0, "reward": 4.603271484375, "reward_std": 7.740581035614014, "rewards/rm_reward_func/mean": 4.603271484375, "rewards/rm_reward_func/std": 18.76112937927246, "step": 828 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 470.0, "completions/mean_length": 340.15625, "completions/mean_terminated_length": 282.875, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "epoch": 0.6632, "grad_norm": 9.52800464630127, "kl": 2.73828125, "learning_rate": 1e-06, "loss": -0.0569, "num_tokens": 11009467.0, "reward": 3.36962890625, "reward_std": 10.40433120727539, "rewards/rm_reward_func/mean": 3.36962890625, "rewards/rm_reward_func/std": 13.418220520019531, "step": 829 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 240.65625, "completions/mean_terminated_length": 222.56668090820312, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "epoch": 0.664, "grad_norm": 10.561025619506836, "kl": 2.5478515625, "learning_rate": 1e-06, "loss": 0.0396, "num_tokens": 11020232.0, "reward": 8.3240966796875, "reward_std": 10.61776351928711, "rewards/rm_reward_func/mean": 8.3240966796875, "rewards/rm_reward_func/std": 14.970850944519043, "step": 830 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 474.0, "completions/mean_length": 330.1875, "completions/mean_terminated_length": 279.2799987792969, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "epoch": 0.6648, "grad_norm": 11.870400428771973, "kl": 3.390625, "learning_rate": 1e-06, "loss": 0.0118, "num_tokens": 11033358.0, "reward": 5.705322265625, "reward_std": 7.648243427276611, "rewards/rm_reward_func/mean": 5.705322265625, "rewards/rm_reward_func/std": 19.251300811767578, "step": 831 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 383.34375, "completions/mean_terminated_length": 370.03448486328125, "completions/min_length": 145.0, "completions/min_terminated_length": 145.0, "epoch": 0.6656, "grad_norm": 7.473958969116211, "kl": 0.41357421875, "learning_rate": 1e-06, "loss": 0.0041, "num_tokens": 11050633.0, "reward": 9.1138916015625, "reward_std": 4.864926338195801, "rewards/rm_reward_func/mean": 9.1138916015625, "rewards/rm_reward_func/std": 15.038235664367676, "step": 832 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 295.90625, "completions/mean_terminated_length": 288.93548583984375, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.6664, "grad_norm": 8.113051414489746, "kl": 1.4638671875, "learning_rate": 1e-06, "loss": 0.0144, "num_tokens": 11063022.0, "reward": 3.6669921875, "reward_std": 7.224357604980469, "rewards/rm_reward_func/mean": 3.6669921875, "rewards/rm_reward_func/std": 12.741311073303223, "step": 833 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 336.53125, "completions/mean_terminated_length": 330.8709716796875, "completions/min_length": 180.0, "completions/min_terminated_length": 180.0, "epoch": 0.6672, "grad_norm": 6.7608513832092285, "kl": 0.941162109375, "learning_rate": 1e-06, "loss": -0.0642, "num_tokens": 11076839.0, "reward": 0.627685546875, "reward_std": 4.866849422454834, "rewards/rm_reward_func/mean": 0.627685546875, "rewards/rm_reward_func/std": 9.038817405700684, "step": 834 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 283.34375, "completions/mean_terminated_length": 268.1000061035156, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "epoch": 0.668, "grad_norm": 22.115230560302734, "kl": 7.388671875, "learning_rate": 1e-06, "loss": 0.1979, "num_tokens": 11091042.0, "reward": 0.4773712158203125, "reward_std": 9.4666109085083, "rewards/rm_reward_func/mean": 0.4773712158203125, "rewards/rm_reward_func/std": 18.32492446899414, "step": 835 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 397.0, "completions/mean_length": 184.3125, "completions/mean_terminated_length": 162.4666748046875, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "epoch": 0.6688, "grad_norm": 31.762920379638672, "kl": 12.359375, "learning_rate": 1e-06, "loss": 0.5706, "num_tokens": 11102388.0, "reward": -8.64892578125, "reward_std": 7.599771499633789, "rewards/rm_reward_func/mean": -8.64892578125, "rewards/rm_reward_func/std": 9.485981941223145, "step": 836 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 424.0, "completions/mean_length": 242.21875, "completions/mean_terminated_length": 192.25926208496094, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "epoch": 0.6696, "grad_norm": 27.026330947875977, "kl": 9.296875, "learning_rate": 1e-06, "loss": 0.2068, "num_tokens": 11115539.0, "reward": -7.0478515625, "reward_std": 10.2100248336792, "rewards/rm_reward_func/mean": -7.0478515625, "rewards/rm_reward_func/std": 18.10556411743164, "step": 837 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 494.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 272.75, "completions/mean_terminated_length": 272.75, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "epoch": 0.6704, "grad_norm": 8.460334777832031, "kl": 2.99609375, "learning_rate": 1e-06, "loss": -0.0436, "num_tokens": 11126803.0, "reward": 3.97998046875, "reward_std": 8.513212203979492, "rewards/rm_reward_func/mean": 3.97998046875, "rewards/rm_reward_func/std": 12.565332412719727, "step": 838 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 388.0, "completions/max_terminated_length": 388.0, "completions/mean_length": 158.90625, "completions/mean_terminated_length": 158.90625, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "epoch": 0.6712, "grad_norm": 7.5811028480529785, "kl": 0.560546875, "learning_rate": 1e-06, "loss": 0.0078, "num_tokens": 11135056.0, "reward": 10.41888427734375, "reward_std": 2.0775063037872314, "rewards/rm_reward_func/mean": 10.41888427734375, "rewards/rm_reward_func/std": 4.920321941375732, "step": 839 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 481.0, "completions/mean_length": 194.75, "completions/mean_terminated_length": 173.60000610351562, "completions/min_length": 97.0, "completions/min_terminated_length": 97.0, "epoch": 0.672, "grad_norm": 17.789752960205078, "kl": 7.279296875, "learning_rate": 1e-06, "loss": 0.2592, "num_tokens": 11143560.0, "reward": -6.46221923828125, "reward_std": 8.128403663635254, "rewards/rm_reward_func/mean": -6.46221923828125, "rewards/rm_reward_func/std": 8.697508811950684, "step": 840 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 324.78125, "completions/mean_terminated_length": 318.7419128417969, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "epoch": 0.6728, "grad_norm": 7.485666751861572, "kl": 1.71435546875, "learning_rate": 1e-06, "loss": 0.0202, "num_tokens": 11158129.0, "reward": 14.0751953125, "reward_std": 8.069329261779785, "rewards/rm_reward_func/mean": 14.0751953125, "rewards/rm_reward_func/std": 10.917937278747559, "step": 841 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 448.0, "completions/max_terminated_length": 448.0, "completions/mean_length": 282.53125, "completions/mean_terminated_length": 282.53125, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "epoch": 0.6736, "grad_norm": 8.429314613342285, "kl": 3.7744140625, "learning_rate": 1e-06, "loss": 0.1521, "num_tokens": 11170770.0, "reward": 8.86328125, "reward_std": 12.041173934936523, "rewards/rm_reward_func/mean": 8.86328125, "rewards/rm_reward_func/std": 16.136621475219727, "step": 842 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 335.0, "completions/mean_length": 147.4375, "completions/mean_terminated_length": 109.72413635253906, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "epoch": 0.6744, "grad_norm": 7.667215347290039, "kl": 2.34375, "learning_rate": 1e-06, "loss": 0.0167, "num_tokens": 11182616.0, "reward": 9.590576171875, "reward_std": 2.2192811965942383, "rewards/rm_reward_func/mean": 9.590576171875, "rewards/rm_reward_func/std": 11.049510955810547, "step": 843 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 389.0, "completions/mean_length": 146.96875, "completions/mean_terminated_length": 135.19354248046875, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "epoch": 0.6752, "grad_norm": 28.290157318115234, "kl": 6.166015625, "learning_rate": 1e-06, "loss": 0.2478, "num_tokens": 11192607.0, "reward": -7.3682861328125, "reward_std": 4.421894073486328, "rewards/rm_reward_func/mean": -7.3682861328125, "rewards/rm_reward_func/std": 13.098028182983398, "step": 844 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 437.0, "completions/mean_length": 278.875, "completions/mean_terminated_length": 271.3548278808594, "completions/min_length": 147.0, "completions/min_terminated_length": 147.0, "epoch": 0.676, "grad_norm": 18.575132369995117, "kl": 5.462890625, "learning_rate": 1e-06, "loss": 0.1575, "num_tokens": 11204379.0, "reward": -2.055908203125, "reward_std": 6.175776481628418, "rewards/rm_reward_func/mean": -2.055908203125, "rewards/rm_reward_func/std": 10.646092414855957, "step": 845 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 335.0, "completions/max_terminated_length": 335.0, "completions/mean_length": 123.96875, "completions/mean_terminated_length": 123.96875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "epoch": 0.6768, "grad_norm": 14.692407608032227, "kl": 4.3916015625, "learning_rate": 1e-06, "loss": 0.1037, "num_tokens": 11213498.0, "reward": 2.0, "reward_std": 3.80605149269104, "rewards/rm_reward_func/mean": 2.0, "rewards/rm_reward_func/std": 9.591236114501953, "step": 846 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 285.75, "completions/mean_terminated_length": 278.45159912109375, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "epoch": 0.6776, "grad_norm": 9.11420726776123, "kl": 1.35791015625, "learning_rate": 1e-06, "loss": -0.1391, "num_tokens": 11224962.0, "reward": 7.869140625, "reward_std": 11.650513648986816, "rewards/rm_reward_func/mean": 7.869140625, "rewards/rm_reward_func/std": 14.315463066101074, "step": 847 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 411.0, "completions/max_terminated_length": 411.0, "completions/mean_length": 196.5625, "completions/mean_terminated_length": 196.5625, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.6784, "grad_norm": 9.939024925231934, "kl": 0.9248046875, "learning_rate": 1e-06, "loss": -0.0194, "num_tokens": 11233572.0, "reward": 7.11181640625, "reward_std": 5.81515645980835, "rewards/rm_reward_func/mean": 7.11181640625, "rewards/rm_reward_func/std": 10.113432884216309, "step": 848 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 374.0, "completions/max_terminated_length": 374.0, "completions/mean_length": 218.65625, "completions/mean_terminated_length": 218.65625, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "epoch": 0.6792, "grad_norm": 14.37784194946289, "kl": 2.833984375, "learning_rate": 1e-06, "loss": -0.0176, "num_tokens": 11242921.0, "reward": -6.9840850830078125, "reward_std": 5.020229816436768, "rewards/rm_reward_func/mean": -6.9840850830078125, "rewards/rm_reward_func/std": 8.650106430053711, "step": 849 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 252.0, "completions/max_terminated_length": 252.0, "completions/mean_length": 133.5, "completions/mean_terminated_length": 133.5, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "epoch": 0.68, "grad_norm": 27.497337341308594, "kl": 2.4345703125, "learning_rate": 1e-06, "loss": 0.0773, "num_tokens": 11250937.0, "reward": 6.5244140625, "reward_std": 7.013378143310547, "rewards/rm_reward_func/mean": 6.5244140625, "rewards/rm_reward_func/std": 11.999058723449707, "step": 850 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 397.40625, "completions/mean_terminated_length": 381.0357360839844, "completions/min_length": 279.0, "completions/min_terminated_length": 279.0, "epoch": 0.6808, "grad_norm": 6.854437351226807, "kl": 0.42431640625, "learning_rate": 1e-06, "loss": -0.0064, "num_tokens": 11266398.0, "reward": 6.41265869140625, "reward_std": 3.7932474613189697, "rewards/rm_reward_func/mean": 6.41265869140625, "rewards/rm_reward_func/std": 17.261632919311523, "step": 851 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 500.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 158.09375, "completions/mean_terminated_length": 158.09375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "epoch": 0.6816, "grad_norm": 10.595882415771484, "kl": 3.65625, "learning_rate": 1e-06, "loss": 0.0449, "num_tokens": 11274801.0, "reward": -12.980224609375, "reward_std": 5.851085662841797, "rewards/rm_reward_func/mean": -12.980224609375, "rewards/rm_reward_func/std": 13.411259651184082, "step": 852 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 232.75, "completions/mean_terminated_length": 223.74192810058594, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "epoch": 0.6824, "grad_norm": 11.931048393249512, "kl": 4.76171875, "learning_rate": 1e-06, "loss": 0.1431, "num_tokens": 11284161.0, "reward": -9.20654296875, "reward_std": 10.515345573425293, "rewards/rm_reward_func/mean": -9.20654296875, "rewards/rm_reward_func/std": 18.860841751098633, "step": 853 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 253.5625, "completions/mean_terminated_length": 236.33334350585938, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "epoch": 0.6832, "grad_norm": 13.002371788024902, "kl": 1.6796875, "learning_rate": 1e-06, "loss": 0.0015, "num_tokens": 11294715.0, "reward": 5.0428466796875, "reward_std": 6.735762596130371, "rewards/rm_reward_func/mean": 5.0428466796875, "rewards/rm_reward_func/std": 12.883460998535156, "step": 854 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 482.0, "completions/mean_length": 225.5625, "completions/mean_terminated_length": 206.4666748046875, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "epoch": 0.684, "grad_norm": 15.005594253540039, "kl": 2.9814453125, "learning_rate": 1e-06, "loss": -0.0264, "num_tokens": 11304629.0, "reward": -3.63232421875, "reward_std": 7.6953206062316895, "rewards/rm_reward_func/mean": -3.63232421875, "rewards/rm_reward_func/std": 14.550228118896484, "step": 855 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 424.0, "completions/mean_length": 222.5625, "completions/mean_terminated_length": 155.7692413330078, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "epoch": 0.6848, "grad_norm": 13.886627197265625, "kl": 2.037109375, "learning_rate": 1e-06, "loss": -0.0007, "num_tokens": 11317831.0, "reward": -3.196533203125, "reward_std": 2.4364099502563477, "rewards/rm_reward_func/mean": -3.196533203125, "rewards/rm_reward_func/std": 9.435223579406738, "step": 856 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 473.0, "completions/max_terminated_length": 473.0, "completions/mean_length": 276.1875, "completions/mean_terminated_length": 276.1875, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "epoch": 0.6856, "grad_norm": 42.20450210571289, "kl": 2.49951171875, "learning_rate": 1e-06, "loss": 0.085, "num_tokens": 11331389.0, "reward": -5.261474609375, "reward_std": 6.263424873352051, "rewards/rm_reward_func/mean": -5.261474609375, "rewards/rm_reward_func/std": 8.193347930908203, "step": 857 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 399.0, "completions/max_terminated_length": 399.0, "completions/mean_length": 256.1875, "completions/mean_terminated_length": 256.1875, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "epoch": 0.6864, "grad_norm": 11.973814010620117, "kl": 0.7158203125, "learning_rate": 1e-06, "loss": 0.0198, "num_tokens": 11341651.0, "reward": 8.2965087890625, "reward_std": 7.567012786865234, "rewards/rm_reward_func/mean": 8.2965087890625, "rewards/rm_reward_func/std": 12.142066955566406, "step": 858 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 231.15625, "completions/mean_terminated_length": 202.10345458984375, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "epoch": 0.6872, "grad_norm": 5.184216499328613, "kl": 1.4384765625, "learning_rate": 1e-06, "loss": 0.0217, "num_tokens": 11351872.0, "reward": 6.35546875, "reward_std": 2.562025785446167, "rewards/rm_reward_func/mean": 6.35546875, "rewards/rm_reward_func/std": 8.541465759277344, "step": 859 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 454.0, "completions/max_terminated_length": 454.0, "completions/mean_length": 188.5625, "completions/mean_terminated_length": 188.5625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "epoch": 0.688, "grad_norm": 8.177552223205566, "kl": 2.0556640625, "learning_rate": 1e-06, "loss": 0.0713, "num_tokens": 11361914.0, "reward": -1.439453125, "reward_std": 3.191701889038086, "rewards/rm_reward_func/mean": -1.439453125, "rewards/rm_reward_func/std": 13.077706336975098, "step": 860 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 401.0, "completions/max_terminated_length": 401.0, "completions/mean_length": 168.21875, "completions/mean_terminated_length": 168.21875, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "epoch": 0.6888, "grad_norm": 12.462203979492188, "kl": 0.489501953125, "learning_rate": 1e-06, "loss": -0.0841, "num_tokens": 11371129.0, "reward": 0.892578125, "reward_std": 6.40195369720459, "rewards/rm_reward_func/mean": 0.892578125, "rewards/rm_reward_func/std": 15.08119010925293, "step": 861 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 496.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 218.28125, "completions/mean_terminated_length": 218.28125, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "epoch": 0.6896, "grad_norm": 9.378438949584961, "kl": 1.27490234375, "learning_rate": 1e-06, "loss": 0.0262, "num_tokens": 11381746.0, "reward": 8.6956787109375, "reward_std": 7.514378547668457, "rewards/rm_reward_func/mean": 8.6956787109375, "rewards/rm_reward_func/std": 12.007038116455078, "step": 862 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 452.0, "completions/max_terminated_length": 452.0, "completions/mean_length": 251.5, "completions/mean_terminated_length": 251.5, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.6904, "grad_norm": 9.017190933227539, "kl": 0.43896484375, "learning_rate": 1e-06, "loss": -0.0491, "num_tokens": 11395986.0, "reward": 4.80322265625, "reward_std": 2.939964771270752, "rewards/rm_reward_func/mean": 4.80322265625, "rewards/rm_reward_func/std": 9.644649505615234, "step": 863 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 288.875, "completions/mean_terminated_length": 281.6773986816406, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "epoch": 0.6912, "grad_norm": 8.287468910217285, "kl": 1.0576171875, "learning_rate": 1e-06, "loss": -0.0984, "num_tokens": 11407550.0, "reward": -3.62890625, "reward_std": 5.023155689239502, "rewards/rm_reward_func/mean": -3.62890625, "rewards/rm_reward_func/std": 15.647568702697754, "step": 864 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 451.0, "completions/mean_length": 234.59375, "completions/mean_terminated_length": 225.64515686035156, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "epoch": 0.692, "grad_norm": 12.379044532775879, "kl": 1.384765625, "learning_rate": 1e-06, "loss": -0.0319, "num_tokens": 11419385.0, "reward": 1.34521484375, "reward_std": 5.130009174346924, "rewards/rm_reward_func/mean": 1.34521484375, "rewards/rm_reward_func/std": 16.12451171875, "step": 865 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 402.0, "completions/mean_length": 298.53125, "completions/mean_terminated_length": 249.2692413330078, "completions/min_length": 94.0, "completions/min_terminated_length": 94.0, "epoch": 0.6928, "grad_norm": 8.24583911895752, "kl": 2.95263671875, "learning_rate": 1e-06, "loss": 0.0369, "num_tokens": 11434746.0, "reward": -10.24560546875, "reward_std": 6.283100128173828, "rewards/rm_reward_func/mean": -10.24560546875, "rewards/rm_reward_func/std": 11.58389949798584, "step": 866 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 402.0, "completions/max_terminated_length": 402.0, "completions/mean_length": 245.84375, "completions/mean_terminated_length": 245.84375, "completions/min_length": 131.0, "completions/min_terminated_length": 131.0, "epoch": 0.6936, "grad_norm": 11.311857223510742, "kl": 1.7294921875, "learning_rate": 1e-06, "loss": 0.0011, "num_tokens": 11445405.0, "reward": -2.562255859375, "reward_std": 4.00094747543335, "rewards/rm_reward_func/mean": -2.562255859375, "rewards/rm_reward_func/std": 5.972076416015625, "step": 867 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 219.96875, "completions/mean_terminated_length": 189.7586212158203, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "epoch": 0.6944, "grad_norm": 21.208097457885742, "kl": 4.55859375, "learning_rate": 1e-06, "loss": 0.1824, "num_tokens": 11455276.0, "reward": 0.225830078125, "reward_std": 9.15473461151123, "rewards/rm_reward_func/mean": 0.225830078125, "rewards/rm_reward_func/std": 12.977889060974121, "step": 868 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 479.0, "completions/mean_length": 295.71875, "completions/mean_terminated_length": 255.6666717529297, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "epoch": 0.6952, "grad_norm": 61.403076171875, "kl": 2.2958984375, "learning_rate": 1e-06, "loss": 0.0541, "num_tokens": 11466739.0, "reward": -4.762786865234375, "reward_std": 12.194247245788574, "rewards/rm_reward_func/mean": -4.762786865234375, "rewards/rm_reward_func/std": 13.474750518798828, "step": 869 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 387.0, "completions/max_terminated_length": 387.0, "completions/mean_length": 194.40625, "completions/mean_terminated_length": 194.40625, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "epoch": 0.696, "grad_norm": 11.072949409484863, "kl": 2.7587890625, "learning_rate": 1e-06, "loss": 0.1886, "num_tokens": 11477680.0, "reward": -2.43963623046875, "reward_std": 2.318965435028076, "rewards/rm_reward_func/mean": -2.43963623046875, "rewards/rm_reward_func/std": 8.259891510009766, "step": 870 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 359.0, "completions/max_terminated_length": 359.0, "completions/mean_length": 120.15625, "completions/mean_terminated_length": 120.15625, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "epoch": 0.6968, "grad_norm": 22.81490135192871, "kl": 2.748046875, "learning_rate": 1e-06, "loss": 0.1335, "num_tokens": 11484381.0, "reward": -4.06182861328125, "reward_std": 3.879387855529785, "rewards/rm_reward_func/mean": -4.06182861328125, "rewards/rm_reward_func/std": 8.522046089172363, "step": 871 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 470.0, "completions/max_terminated_length": 470.0, "completions/mean_length": 184.25, "completions/mean_terminated_length": 184.25, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "epoch": 0.6976, "grad_norm": 21.19168472290039, "kl": 1.7607421875, "learning_rate": 1e-06, "loss": 0.0231, "num_tokens": 11493045.0, "reward": -6.307861328125, "reward_std": 5.177760601043701, "rewards/rm_reward_func/mean": -6.307861328125, "rewards/rm_reward_func/std": 17.660837173461914, "step": 872 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 491.0, "completions/max_terminated_length": 491.0, "completions/mean_length": 302.21875, "completions/mean_terminated_length": 302.21875, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "epoch": 0.6984, "grad_norm": 13.353934288024902, "kl": 0.684326171875, "learning_rate": 1e-06, "loss": -0.0657, "num_tokens": 11505260.0, "reward": 6.2265625, "reward_std": 4.352338790893555, "rewards/rm_reward_func/mean": 6.2265625, "rewards/rm_reward_func/std": 12.638473510742188, "step": 873 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 469.0, "completions/max_terminated_length": 469.0, "completions/mean_length": 275.75, "completions/mean_terminated_length": 275.75, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "epoch": 0.6992, "grad_norm": 10.98495864868164, "kl": 2.4599609375, "learning_rate": 1e-06, "loss": 0.0102, "num_tokens": 11518476.0, "reward": -2.28125, "reward_std": 8.162097930908203, "rewards/rm_reward_func/mean": -2.28125, "rewards/rm_reward_func/std": 16.660022735595703, "step": 874 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 360.09375, "completions/mean_terminated_length": 309.4583435058594, "completions/min_length": 140.0, "completions/min_terminated_length": 140.0, "epoch": 0.7, "grad_norm": 10.327630043029785, "kl": 0.825439453125, "learning_rate": 1e-06, "loss": 0.0071, "num_tokens": 11532407.0, "reward": 5.72357177734375, "reward_std": 6.062471389770508, "rewards/rm_reward_func/mean": 5.72357177734375, "rewards/rm_reward_func/std": 8.450596809387207, "step": 875 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 261.65625, "completions/mean_terminated_length": 253.5806427001953, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "epoch": 0.7008, "grad_norm": 12.771651268005371, "kl": 0.5576171875, "learning_rate": 1e-06, "loss": -0.3164, "num_tokens": 11544620.0, "reward": -0.7434616088867188, "reward_std": 7.011323928833008, "rewards/rm_reward_func/mean": -0.7434616088867188, "rewards/rm_reward_func/std": 9.20909595489502, "step": 876 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 449.0, "completions/mean_length": 357.53125, "completions/mean_terminated_length": 306.04168701171875, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "epoch": 0.7016, "grad_norm": 6.507169246673584, "kl": 0.88671875, "learning_rate": 1e-06, "loss": 0.0546, "num_tokens": 11558797.0, "reward": 2.79498291015625, "reward_std": 4.3798723220825195, "rewards/rm_reward_func/mean": 2.79498291015625, "rewards/rm_reward_func/std": 11.152603149414062, "step": 877 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 459.0, "completions/mean_length": 337.21875, "completions/mean_terminated_length": 325.5666809082031, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "epoch": 0.7024, "grad_norm": 7.747379302978516, "kl": 2.21875, "learning_rate": 1e-06, "loss": -0.0081, "num_tokens": 11572412.0, "reward": -0.7412109375, "reward_std": 10.57952880859375, "rewards/rm_reward_func/mean": -0.7412109375, "rewards/rm_reward_func/std": 18.887123107910156, "step": 878 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 467.0, "completions/mean_length": 245.84375, "completions/mean_terminated_length": 218.3103485107422, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "epoch": 0.7032, "grad_norm": 12.370904922485352, "kl": 1.8740234375, "learning_rate": 1e-06, "loss": 0.0823, "num_tokens": 11583719.0, "reward": -5.319202423095703, "reward_std": 8.250015258789062, "rewards/rm_reward_func/mean": -5.319202423095703, "rewards/rm_reward_func/std": 13.415038108825684, "step": 879 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 256.78125, "completions/mean_terminated_length": 239.7666778564453, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "epoch": 0.704, "grad_norm": 6.51011848449707, "kl": 0.8095703125, "learning_rate": 1e-06, "loss": 0.0124, "num_tokens": 11594712.0, "reward": 12.4833984375, "reward_std": 4.730257034301758, "rewards/rm_reward_func/mean": 12.4833984375, "rewards/rm_reward_func/std": 8.233878135681152, "step": 880 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 372.90625, "completions/mean_terminated_length": 368.4193420410156, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "epoch": 0.7048, "grad_norm": 6.078239440917969, "kl": 0.372802734375, "learning_rate": 1e-06, "loss": -0.0806, "num_tokens": 11608573.0, "reward": 10.44952392578125, "reward_std": 5.383957386016846, "rewards/rm_reward_func/mean": 10.44952392578125, "rewards/rm_reward_func/std": 14.256927490234375, "step": 881 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 276.0625, "completions/mean_terminated_length": 251.65516662597656, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "epoch": 0.7056, "grad_norm": 13.18306827545166, "kl": 3.087646484375, "learning_rate": 1e-06, "loss": 0.2504, "num_tokens": 11622719.0, "reward": 2.23345947265625, "reward_std": 10.839166641235352, "rewards/rm_reward_func/mean": 2.23345947265625, "rewards/rm_reward_func/std": 13.140902519226074, "step": 882 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 174.28125, "completions/mean_terminated_length": 61.708335876464844, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "epoch": 0.7064, "grad_norm": 9.840855598449707, "kl": 0.5, "learning_rate": 1e-06, "loss": -0.0233, "num_tokens": 11631200.0, "reward": 2.555927276611328, "reward_std": 1.7101624011993408, "rewards/rm_reward_func/mean": 2.555927276611328, "rewards/rm_reward_func/std": 6.12503719329834, "step": 883 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 397.53125, "completions/mean_terminated_length": 385.6896667480469, "completions/min_length": 110.0, "completions/min_terminated_length": 110.0, "epoch": 0.7072, "grad_norm": 6.078858852386475, "kl": 0.498046875, "learning_rate": 1e-06, "loss": -0.0905, "num_tokens": 11647873.0, "reward": 2.686767578125, "reward_std": 7.326623439788818, "rewards/rm_reward_func/mean": 2.686767578125, "rewards/rm_reward_func/std": 11.423783302307129, "step": 884 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 481.0, "completions/mean_length": 377.34375, "completions/mean_terminated_length": 339.6399841308594, "completions/min_length": 124.0, "completions/min_terminated_length": 124.0, "epoch": 0.708, "grad_norm": 10.225689888000488, "kl": 2.7353515625, "learning_rate": 1e-06, "loss": 0.0753, "num_tokens": 11662996.0, "reward": 2.117431640625, "reward_std": 9.586727142333984, "rewards/rm_reward_func/mean": 2.117431640625, "rewards/rm_reward_func/std": 12.49357795715332, "step": 885 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 482.0, "completions/mean_length": 235.8125, "completions/mean_terminated_length": 196.35714721679688, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "epoch": 0.7088, "grad_norm": 17.94410514831543, "kl": 3.6484375, "learning_rate": 1e-06, "loss": 0.1815, "num_tokens": 11672782.0, "reward": -1.3177490234375, "reward_std": 9.219121932983398, "rewards/rm_reward_func/mean": -1.3177490234375, "rewards/rm_reward_func/std": 11.736974716186523, "step": 886 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 326.125, "completions/mean_terminated_length": 264.16668701171875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "epoch": 0.7096, "grad_norm": 11.345696449279785, "kl": 5.15625, "learning_rate": 1e-06, "loss": 0.192, "num_tokens": 11690626.0, "reward": 4.96630859375, "reward_std": 6.32305908203125, "rewards/rm_reward_func/mean": 4.96630859375, "rewards/rm_reward_func/std": 13.138326644897461, "step": 887 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 408.0, "completions/max_terminated_length": 408.0, "completions/mean_length": 187.46875, "completions/mean_terminated_length": 187.46875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "epoch": 0.7104, "grad_norm": 21.07513427734375, "kl": 0.44580078125, "learning_rate": 1e-06, "loss": -0.0441, "num_tokens": 11699697.0, "reward": 3.6982421875, "reward_std": 3.48081636428833, "rewards/rm_reward_func/mean": 3.6982421875, "rewards/rm_reward_func/std": 14.544777870178223, "step": 888 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 460.0, "completions/mean_length": 328.15625, "completions/mean_terminated_length": 322.2257995605469, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "epoch": 0.7112, "grad_norm": 7.2959303855896, "kl": 3.615478515625, "learning_rate": 1e-06, "loss": 0.152, "num_tokens": 11712518.0, "reward": 3.5261077880859375, "reward_std": 9.151225090026855, "rewards/rm_reward_func/mean": 3.5261077880859375, "rewards/rm_reward_func/std": 11.036227226257324, "step": 889 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 467.0, "completions/mean_length": 306.21875, "completions/mean_terminated_length": 268.1111145019531, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "epoch": 0.712, "grad_norm": 7.229876518249512, "kl": 4.206298828125, "learning_rate": 1e-06, "loss": 0.1888, "num_tokens": 11726501.0, "reward": 5.666015625, "reward_std": 12.855541229248047, "rewards/rm_reward_func/mean": 5.666015625, "rewards/rm_reward_func/std": 19.8636531829834, "step": 890 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 465.0, "completions/mean_length": 177.25, "completions/mean_terminated_length": 166.4516143798828, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "epoch": 0.7128, "grad_norm": 11.873459815979004, "kl": 4.671875, "learning_rate": 1e-06, "loss": 0.1815, "num_tokens": 11735293.0, "reward": -2.13092041015625, "reward_std": 2.0769262313842773, "rewards/rm_reward_func/mean": -2.13092041015625, "rewards/rm_reward_func/std": 10.466885566711426, "step": 891 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 430.71875, "completions/mean_terminated_length": 375.1052551269531, "completions/min_length": 193.0, "completions/min_terminated_length": 193.0, "epoch": 0.7136, "grad_norm": 15.590665817260742, "kl": 9.9453125, "learning_rate": 1e-06, "loss": 0.3228, "num_tokens": 11751772.0, "reward": -10.29931640625, "reward_std": 10.349671363830566, "rewards/rm_reward_func/mean": -10.29931640625, "rewards/rm_reward_func/std": 14.615714073181152, "step": 892 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 477.0, "completions/mean_length": 215.03125, "completions/mean_terminated_length": 172.60714721679688, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "epoch": 0.7144, "grad_norm": 23.91371726989746, "kl": 7.96875, "learning_rate": 1e-06, "loss": 0.3347, "num_tokens": 11762333.0, "reward": -2.80712890625, "reward_std": 3.1941628456115723, "rewards/rm_reward_func/mean": -2.80712890625, "rewards/rm_reward_func/std": 15.444784164428711, "step": 893 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40625, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 337.59375, "completions/mean_terminated_length": 218.26315307617188, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "epoch": 0.7152, "grad_norm": 15.916248321533203, "kl": 10.64453125, "learning_rate": 1e-06, "loss": 0.4986, "num_tokens": 11777384.0, "reward": -6.3330078125, "reward_std": 8.051085472106934, "rewards/rm_reward_func/mean": -6.3330078125, "rewards/rm_reward_func/std": 13.273330688476562, "step": 894 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 423.625, "completions/mean_terminated_length": 389.0434875488281, "completions/min_length": 141.0, "completions/min_terminated_length": 141.0, "epoch": 0.716, "grad_norm": 27.08871078491211, "kl": 11.65234375, "learning_rate": 1e-06, "loss": 0.4485, "num_tokens": 11792972.0, "reward": -5.36322021484375, "reward_std": 8.532169342041016, "rewards/rm_reward_func/mean": -5.36322021484375, "rewards/rm_reward_func/std": 14.15001106262207, "step": 895 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 454.0, "completions/mean_length": 302.5, "completions/mean_terminated_length": 232.6666717529297, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "epoch": 0.7168, "grad_norm": 15.366225242614746, "kl": 8.49609375, "learning_rate": 1e-06, "loss": 0.2878, "num_tokens": 11804820.0, "reward": -3.9163665771484375, "reward_std": 10.814449310302734, "rewards/rm_reward_func/mean": -3.9163665771484375, "rewards/rm_reward_func/std": 13.626547813415527, "step": 896 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 459.0, "completions/mean_length": 301.5625, "completions/mean_terminated_length": 262.59259033203125, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "epoch": 0.7176, "grad_norm": 16.501802444458008, "kl": 6.5625, "learning_rate": 1e-06, "loss": 0.3279, "num_tokens": 11817710.0, "reward": -13.09716796875, "reward_std": 3.4331793785095215, "rewards/rm_reward_func/mean": -13.09716796875, "rewards/rm_reward_func/std": 7.296655178070068, "step": 897 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 383.15625, "completions/mean_terminated_length": 305.8500061035156, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "epoch": 0.7184, "grad_norm": 53.010684967041016, "kl": 7.34423828125, "learning_rate": 1e-06, "loss": 0.2151, "num_tokens": 11835283.0, "reward": -2.9622802734375, "reward_std": 7.178778648376465, "rewards/rm_reward_func/mean": -2.9622802734375, "rewards/rm_reward_func/std": 15.090096473693848, "step": 898 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 429.0, "completions/mean_length": 188.46875, "completions/mean_terminated_length": 178.03225708007812, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "epoch": 0.7192, "grad_norm": 12.646398544311523, "kl": 2.4697265625, "learning_rate": 1e-06, "loss": 0.0837, "num_tokens": 11844258.0, "reward": -5.5667266845703125, "reward_std": 3.9434971809387207, "rewards/rm_reward_func/mean": -5.5667266845703125, "rewards/rm_reward_func/std": 9.585037231445312, "step": 899 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 479.0, "completions/mean_length": 374.71875, "completions/mean_terminated_length": 336.2799987792969, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "epoch": 0.72, "grad_norm": 14.73233413696289, "kl": 4.44140625, "learning_rate": 1e-06, "loss": 0.2017, "num_tokens": 11858337.0, "reward": 4.2802734375, "reward_std": 4.078566074371338, "rewards/rm_reward_func/mean": 4.2802734375, "rewards/rm_reward_func/std": 18.690641403198242, "step": 900 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 297.0625, "completions/mean_terminated_length": 236.87998962402344, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "epoch": 0.7208, "grad_norm": 12.702629089355469, "kl": 1.1796875, "learning_rate": 1e-06, "loss": 0.0394, "num_tokens": 11871475.0, "reward": -1.2879791259765625, "reward_std": 9.910463333129883, "rewards/rm_reward_func/mean": -1.2879791259765625, "rewards/rm_reward_func/std": 13.349971771240234, "step": 901 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 477.0, "completions/mean_length": 292.21875, "completions/mean_terminated_length": 260.8214416503906, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.7216, "grad_norm": 4.130387306213379, "kl": 1.59130859375, "learning_rate": 1e-06, "loss": 0.0345, "num_tokens": 11884650.0, "reward": -2.958984375, "reward_std": 4.643342971801758, "rewards/rm_reward_func/mean": -2.958984375, "rewards/rm_reward_func/std": 10.972875595092773, "step": 902 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 469.0, "completions/mean_length": 226.59375, "completions/mean_terminated_length": 146.67999267578125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "epoch": 0.7224, "grad_norm": 7.769096851348877, "kl": 3.14453125, "learning_rate": 1e-06, "loss": 0.1549, "num_tokens": 11894645.0, "reward": -5.154296875, "reward_std": 1.5148166418075562, "rewards/rm_reward_func/mean": -5.154296875, "rewards/rm_reward_func/std": 8.897757530212402, "step": 903 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 368.5625, "completions/mean_terminated_length": 335.4615478515625, "completions/min_length": 99.0, "completions/min_terminated_length": 99.0, "epoch": 0.7232, "grad_norm": 8.09273910522461, "kl": 2.3486328125, "learning_rate": 1e-06, "loss": 0.146, "num_tokens": 11908447.0, "reward": -3.269775390625, "reward_std": 5.145322799682617, "rewards/rm_reward_func/mean": -3.269775390625, "rewards/rm_reward_func/std": 9.46324634552002, "step": 904 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 311.125, "completions/mean_terminated_length": 264.76922607421875, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "epoch": 0.724, "grad_norm": 8.956731796264648, "kl": 2.220703125, "learning_rate": 1e-06, "loss": 0.0198, "num_tokens": 11921539.0, "reward": -2.416015625, "reward_std": 7.836517333984375, "rewards/rm_reward_func/mean": -2.416015625, "rewards/rm_reward_func/std": 19.40036392211914, "step": 905 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 491.0, "completions/mean_length": 403.78125, "completions/mean_terminated_length": 354.5909118652344, "completions/min_length": 117.0, "completions/min_terminated_length": 117.0, "epoch": 0.7248, "grad_norm": 5.295882701873779, "kl": 3.423828125, "learning_rate": 1e-06, "loss": 0.1755, "num_tokens": 11936668.0, "reward": -7.770965576171875, "reward_std": 7.4591522216796875, "rewards/rm_reward_func/mean": -7.770965576171875, "rewards/rm_reward_func/std": 9.990768432617188, "step": 906 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.4375, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 367.15625, "completions/mean_terminated_length": 254.5, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.7256, "grad_norm": 9.633946418762207, "kl": 2.376953125, "learning_rate": 1e-06, "loss": 0.0554, "num_tokens": 11952129.0, "reward": -9.96728515625, "reward_std": 5.492212772369385, "rewards/rm_reward_func/mean": -9.96728515625, "rewards/rm_reward_func/std": 7.280327320098877, "step": 907 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 474.0, "completions/mean_length": 175.5, "completions/mean_terminated_length": 127.42857360839844, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "epoch": 0.7264, "grad_norm": 4.560550212860107, "kl": 0.552734375, "learning_rate": 1e-06, "loss": 0.0122, "num_tokens": 11962097.0, "reward": 7.42626953125, "reward_std": 3.087636709213257, "rewards/rm_reward_func/mean": 7.42626953125, "rewards/rm_reward_func/std": 7.106700897216797, "step": 908 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 299.21875, "completions/mean_terminated_length": 285.0333557128906, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "epoch": 0.7272, "grad_norm": 7.450979232788086, "kl": 1.3125, "learning_rate": 1e-06, "loss": 0.0609, "num_tokens": 11974520.0, "reward": 3.33935546875, "reward_std": 8.265787124633789, "rewards/rm_reward_func/mean": 3.33935546875, "rewards/rm_reward_func/std": 13.333514213562012, "step": 909 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 366.6875, "completions/mean_terminated_length": 309.8260803222656, "completions/min_length": 104.0, "completions/min_terminated_length": 104.0, "epoch": 0.728, "grad_norm": 8.965705871582031, "kl": 1.3427734375, "learning_rate": 1e-06, "loss": 0.0475, "num_tokens": 11989574.0, "reward": -3.5350265502929688, "reward_std": 7.181827068328857, "rewards/rm_reward_func/mean": -3.5350265502929688, "rewards/rm_reward_func/std": 8.107301712036133, "step": 910 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 476.0, "completions/mean_length": 325.25, "completions/mean_terminated_length": 290.6666564941406, "completions/min_length": 124.0, "completions/min_terminated_length": 124.0, "epoch": 0.7288, "grad_norm": 6.053767681121826, "kl": 0.9969482421875, "learning_rate": 1e-06, "loss": 0.0931, "num_tokens": 12003822.0, "reward": -4.5394287109375, "reward_std": 6.439982891082764, "rewards/rm_reward_func/mean": -4.5394287109375, "rewards/rm_reward_func/std": 12.76736831665039, "step": 911 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.46875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 380.8125, "completions/mean_terminated_length": 265.058837890625, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "epoch": 0.7296, "grad_norm": 13.645872116088867, "kl": 1.7490234375, "learning_rate": 1e-06, "loss": 0.0605, "num_tokens": 12019832.0, "reward": -1.08984375, "reward_std": 7.622467994689941, "rewards/rm_reward_func/mean": -1.08984375, "rewards/rm_reward_func/std": 14.811065673828125, "step": 912 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 312.78125, "completions/mean_terminated_length": 266.8077087402344, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "epoch": 0.7304, "grad_norm": 9.30956745147705, "kl": 1.2353515625, "learning_rate": 1e-06, "loss": -0.0088, "num_tokens": 12032193.0, "reward": 1.819305419921875, "reward_std": 11.730962753295898, "rewards/rm_reward_func/mean": 1.819305419921875, "rewards/rm_reward_func/std": 17.3555908203125, "step": 913 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 403.78125, "completions/mean_terminated_length": 367.7083435058594, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "epoch": 0.7312, "grad_norm": 8.707047462463379, "kl": 1.447509765625, "learning_rate": 1e-06, "loss": -0.0823, "num_tokens": 12047786.0, "reward": 1.6616268157958984, "reward_std": 7.0725531578063965, "rewards/rm_reward_func/mean": 1.6616268157958984, "rewards/rm_reward_func/std": 11.841200828552246, "step": 914 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 325.09375, "completions/mean_terminated_length": 305.75860595703125, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.732, "grad_norm": 6.918102741241455, "kl": 1.4365234375, "learning_rate": 1e-06, "loss": 0.0435, "num_tokens": 12061229.0, "reward": -0.952178955078125, "reward_std": 10.511100769042969, "rewards/rm_reward_func/mean": -0.952178955078125, "rewards/rm_reward_func/std": 13.577516555786133, "step": 915 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 302.4375, "completions/mean_terminated_length": 272.5, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "epoch": 0.7328, "grad_norm": 16.212295532226562, "kl": 1.50927734375, "learning_rate": 1e-06, "loss": 0.2198, "num_tokens": 12073931.0, "reward": -2.2662353515625, "reward_std": 6.384040832519531, "rewards/rm_reward_func/mean": -2.2662353515625, "rewards/rm_reward_func/std": 8.968281745910645, "step": 916 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 220.5625, "completions/mean_terminated_length": 201.1333465576172, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "epoch": 0.7336, "grad_norm": 6.973321914672852, "kl": 0.521484375, "learning_rate": 1e-06, "loss": -0.0016, "num_tokens": 12084549.0, "reward": 6.8271484375, "reward_std": 3.8583192825317383, "rewards/rm_reward_func/mean": 6.8271484375, "rewards/rm_reward_func/std": 11.556774139404297, "step": 917 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 432.0, "completions/max_terminated_length": 432.0, "completions/mean_length": 193.75, "completions/mean_terminated_length": 193.75, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "epoch": 0.7344, "grad_norm": 11.093525886535645, "kl": 0.890625, "learning_rate": 1e-06, "loss": -0.0353, "num_tokens": 12093829.0, "reward": 7.5962066650390625, "reward_std": 5.227684497833252, "rewards/rm_reward_func/mean": 7.5962066650390625, "rewards/rm_reward_func/std": 7.640827178955078, "step": 918 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 263.03125, "completions/mean_terminated_length": 246.433349609375, "completions/min_length": 109.0, "completions/min_terminated_length": 109.0, "epoch": 0.7352, "grad_norm": 8.780437469482422, "kl": 1.28955078125, "learning_rate": 1e-06, "loss": -0.0192, "num_tokens": 12105222.0, "reward": 0.42479705810546875, "reward_std": 5.950567722320557, "rewards/rm_reward_func/mean": 0.42479705810546875, "rewards/rm_reward_func/std": 10.092032432556152, "step": 919 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 482.0, "completions/mean_length": 341.21875, "completions/mean_terminated_length": 284.29168701171875, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "epoch": 0.736, "grad_norm": 18.315034866333008, "kl": 2.19921875, "learning_rate": 1e-06, "loss": 0.3037, "num_tokens": 12118829.0, "reward": 7.9627838134765625, "reward_std": 6.95778226852417, "rewards/rm_reward_func/mean": 7.9627838134765625, "rewards/rm_reward_func/std": 21.903451919555664, "step": 920 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 388.34375, "completions/mean_terminated_length": 339.95654296875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "epoch": 0.7368, "grad_norm": 10.440993309020996, "kl": 0.66943359375, "learning_rate": 1e-06, "loss": 0.1228, "num_tokens": 12133104.0, "reward": 8.855255126953125, "reward_std": 10.921690940856934, "rewards/rm_reward_func/mean": 8.855255126953125, "rewards/rm_reward_func/std": 17.32367515563965, "step": 921 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 476.0, "completions/mean_length": 327.875, "completions/mean_terminated_length": 285.3846130371094, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "epoch": 0.7376, "grad_norm": 13.36120891571045, "kl": 1.40380859375, "learning_rate": 1e-06, "loss": -0.0399, "num_tokens": 12145796.0, "reward": -1.751861572265625, "reward_std": 5.661749839782715, "rewards/rm_reward_func/mean": -1.751861572265625, "rewards/rm_reward_func/std": 7.456704616546631, "step": 922 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 396.90625, "completions/mean_terminated_length": 370.3461608886719, "completions/min_length": 187.0, "completions/min_terminated_length": 187.0, "epoch": 0.7384, "grad_norm": 7.57150936126709, "kl": 4.10546875, "learning_rate": 1e-06, "loss": 0.0999, "num_tokens": 12160545.0, "reward": 2.98095703125, "reward_std": 12.732685089111328, "rewards/rm_reward_func/mean": 2.98095703125, "rewards/rm_reward_func/std": 16.23927879333496, "step": 923 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 403.375, "completions/mean_terminated_length": 360.86956787109375, "completions/min_length": 141.0, "completions/min_terminated_length": 141.0, "epoch": 0.7392, "grad_norm": 17.749086380004883, "kl": 3.669921875, "learning_rate": 1e-06, "loss": 0.1162, "num_tokens": 12175453.0, "reward": -4.994476318359375, "reward_std": 6.3849382400512695, "rewards/rm_reward_func/mean": -4.994476318359375, "rewards/rm_reward_func/std": 7.732586860656738, "step": 924 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 375.03125, "completions/mean_terminated_length": 349.6666564941406, "completions/min_length": 159.0, "completions/min_terminated_length": 159.0, "epoch": 0.74, "grad_norm": 10.675846099853516, "kl": 2.8515625, "learning_rate": 1e-06, "loss": 0.0318, "num_tokens": 12189734.0, "reward": -0.42730712890625, "reward_std": 8.244186401367188, "rewards/rm_reward_func/mean": -0.42730712890625, "rewards/rm_reward_func/std": 11.274063110351562, "step": 925 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 321.875, "completions/mean_terminated_length": 258.5, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "epoch": 0.7408, "grad_norm": 20.539968490600586, "kl": 7.5546875, "learning_rate": 1e-06, "loss": 0.1656, "num_tokens": 12202114.0, "reward": -17.49755859375, "reward_std": 5.544781684875488, "rewards/rm_reward_func/mean": -17.49755859375, "rewards/rm_reward_func/std": 8.136129379272461, "step": 926 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 215.375, "completions/mean_terminated_length": 146.92308044433594, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "epoch": 0.7416, "grad_norm": 16.78243637084961, "kl": 1.171875, "learning_rate": 1e-06, "loss": 0.0322, "num_tokens": 12214726.0, "reward": -3.51953125, "reward_std": 2.579955577850342, "rewards/rm_reward_func/mean": -3.51953125, "rewards/rm_reward_func/std": 12.52515697479248, "step": 927 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 469.0, "completions/mean_length": 340.34375, "completions/mean_terminated_length": 292.2799987792969, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "epoch": 0.7424, "grad_norm": 34.499900817871094, "kl": 9.70751953125, "learning_rate": 1e-06, "loss": 0.3566, "num_tokens": 12227649.0, "reward": -9.71435546875, "reward_std": 4.9702467918396, "rewards/rm_reward_func/mean": -9.71435546875, "rewards/rm_reward_func/std": 13.78485107421875, "step": 928 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 340.21875, "completions/mean_terminated_length": 308.40740966796875, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "epoch": 0.7432, "grad_norm": 6.942431449890137, "kl": 4.47021484375, "learning_rate": 1e-06, "loss": 0.0818, "num_tokens": 12240512.0, "reward": -4.043701171875, "reward_std": 9.750587463378906, "rewards/rm_reward_func/mean": -4.043701171875, "rewards/rm_reward_func/std": 10.625539779663086, "step": 929 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 423.0, "completions/mean_length": 287.4375, "completions/mean_terminated_length": 212.58334350585938, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "epoch": 0.744, "grad_norm": 31.259197235107422, "kl": 9.36083984375, "learning_rate": 1e-06, "loss": 0.3652, "num_tokens": 12253814.0, "reward": -10.76513671875, "reward_std": 2.81522274017334, "rewards/rm_reward_func/mean": -10.76513671875, "rewards/rm_reward_func/std": 10.83386516571045, "step": 930 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 469.0, "completions/mean_length": 214.125, "completions/mean_terminated_length": 194.2666778564453, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "epoch": 0.7448, "grad_norm": 33.200462341308594, "kl": 10.28125, "learning_rate": 1e-06, "loss": 0.465, "num_tokens": 12265218.0, "reward": -19.12255859375, "reward_std": 3.9576473236083984, "rewards/rm_reward_func/mean": -19.12255859375, "rewards/rm_reward_func/std": 7.913651466369629, "step": 931 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 276.65625, "completions/mean_terminated_length": 269.06451416015625, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "epoch": 0.7456, "grad_norm": 22.295116424560547, "kl": 6.87109375, "learning_rate": 1e-06, "loss": 0.1678, "num_tokens": 12277919.0, "reward": -6.76416015625, "reward_std": 9.437423706054688, "rewards/rm_reward_func/mean": -6.76416015625, "rewards/rm_reward_func/std": 13.566336631774902, "step": 932 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 321.875, "completions/mean_terminated_length": 294.71429443359375, "completions/min_length": 116.0, "completions/min_terminated_length": 116.0, "epoch": 0.7464, "grad_norm": 38.953163146972656, "kl": 10.8125, "learning_rate": 1e-06, "loss": 0.3992, "num_tokens": 12290051.0, "reward": -13.24609375, "reward_std": 5.0346527099609375, "rewards/rm_reward_func/mean": -13.24609375, "rewards/rm_reward_func/std": 5.71676778793335, "step": 933 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 240.125, "completions/mean_terminated_length": 231.35482788085938, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "epoch": 0.7472, "grad_norm": 16.947662353515625, "kl": 5.8046875, "learning_rate": 1e-06, "loss": 0.1648, "num_tokens": 12299375.0, "reward": -9.80364990234375, "reward_std": 7.922030448913574, "rewards/rm_reward_func/mean": -9.80364990234375, "rewards/rm_reward_func/std": 8.047533988952637, "step": 934 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 493.0, "completions/mean_length": 334.5, "completions/mean_terminated_length": 265.0434875488281, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "epoch": 0.748, "grad_norm": 20.085559844970703, "kl": 3.55517578125, "learning_rate": 1e-06, "loss": 0.0963, "num_tokens": 12313439.0, "reward": -4.38671875, "reward_std": 5.369791030883789, "rewards/rm_reward_func/mean": -4.38671875, "rewards/rm_reward_func/std": 12.364655494689941, "step": 935 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 478.0, "completions/max_terminated_length": 478.0, "completions/mean_length": 190.5625, "completions/mean_terminated_length": 190.5625, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "epoch": 0.7488, "grad_norm": 19.231365203857422, "kl": 5.12109375, "learning_rate": 1e-06, "loss": 0.3493, "num_tokens": 12322225.0, "reward": -9.181640625, "reward_std": 6.0393853187561035, "rewards/rm_reward_func/mean": -9.181640625, "rewards/rm_reward_func/std": 11.844439506530762, "step": 936 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 376.625, "completions/mean_terminated_length": 331.5, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "epoch": 0.7496, "grad_norm": 12.45418930053711, "kl": 6.169921875, "learning_rate": 1e-06, "loss": 0.336, "num_tokens": 12338125.0, "reward": -4.42236328125, "reward_std": 8.240564346313477, "rewards/rm_reward_func/mean": -4.42236328125, "rewards/rm_reward_func/std": 17.537168502807617, "step": 937 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 455.0, "completions/mean_length": 244.09375, "completions/mean_terminated_length": 216.37930297851562, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "epoch": 0.7504, "grad_norm": 15.027655601501465, "kl": 2.1923828125, "learning_rate": 1e-06, "loss": 0.0436, "num_tokens": 12351256.0, "reward": -4.74468994140625, "reward_std": 5.563222885131836, "rewards/rm_reward_func/mean": -4.74468994140625, "rewards/rm_reward_func/std": 10.182250022888184, "step": 938 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 455.0, "completions/mean_length": 273.71875, "completions/mean_terminated_length": 249.0689697265625, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "epoch": 0.7512, "grad_norm": 6.310357570648193, "kl": 2.48291015625, "learning_rate": 1e-06, "loss": 0.1097, "num_tokens": 12362543.0, "reward": -3.814697265625, "reward_std": 5.3505682945251465, "rewards/rm_reward_func/mean": -3.814697265625, "rewards/rm_reward_func/std": 7.993979454040527, "step": 939 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 448.0, "completions/mean_length": 170.59375, "completions/mean_terminated_length": 147.83334350585938, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.752, "grad_norm": 10.239607810974121, "kl": 1.9765625, "learning_rate": 1e-06, "loss": 0.2201, "num_tokens": 12370810.0, "reward": 6.2581787109375, "reward_std": 5.198629379272461, "rewards/rm_reward_func/mean": 6.2581787109375, "rewards/rm_reward_func/std": 11.37491512298584, "step": 940 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 481.0, "completions/mean_length": 358.1875, "completions/mean_terminated_length": 342.2758483886719, "completions/min_length": 85.0, "completions/min_terminated_length": 85.0, "epoch": 0.7528, "grad_norm": 6.657660961151123, "kl": 2.30615234375, "learning_rate": 1e-06, "loss": 0.0582, "num_tokens": 12384584.0, "reward": 1.087982177734375, "reward_std": 10.667032241821289, "rewards/rm_reward_func/mean": 1.087982177734375, "rewards/rm_reward_func/std": 15.517912864685059, "step": 941 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 487.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 243.21875, "completions/mean_terminated_length": 243.21875, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "epoch": 0.7536, "grad_norm": 7.709587574005127, "kl": 2.3125, "learning_rate": 1e-06, "loss": 0.123, "num_tokens": 12396471.0, "reward": -9.607421875, "reward_std": 2.2891464233398438, "rewards/rm_reward_func/mean": -9.607421875, "rewards/rm_reward_func/std": 11.87063217163086, "step": 942 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 330.40625, "completions/mean_terminated_length": 296.77777099609375, "completions/min_length": 124.0, "completions/min_terminated_length": 124.0, "epoch": 0.7544, "grad_norm": 9.007935523986816, "kl": 0.546142578125, "learning_rate": 1e-06, "loss": 0.0883, "num_tokens": 12412252.0, "reward": 3.67486572265625, "reward_std": 3.7311360836029053, "rewards/rm_reward_func/mean": 3.67486572265625, "rewards/rm_reward_func/std": 8.098952293395996, "step": 943 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 447.0, "completions/max_terminated_length": 447.0, "completions/mean_length": 180.96875, "completions/mean_terminated_length": 180.96875, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "epoch": 0.7552, "grad_norm": 7.097393989562988, "kl": 1.11328125, "learning_rate": 1e-06, "loss": 0.1134, "num_tokens": 12421491.0, "reward": 3.06976318359375, "reward_std": 3.404435873031616, "rewards/rm_reward_func/mean": 3.06976318359375, "rewards/rm_reward_func/std": 5.2598958015441895, "step": 944 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 474.0, "completions/mean_length": 225.625, "completions/mean_terminated_length": 206.53334045410156, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "epoch": 0.756, "grad_norm": 6.910060405731201, "kl": 0.96484375, "learning_rate": 1e-06, "loss": -0.0059, "num_tokens": 12432391.0, "reward": 1.87451171875, "reward_std": 6.181256294250488, "rewards/rm_reward_func/mean": 1.87451171875, "rewards/rm_reward_func/std": 9.241499900817871, "step": 945 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 264.9375, "completions/mean_terminated_length": 168.26087951660156, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "epoch": 0.7568, "grad_norm": 18.712726593017578, "kl": 0.5518798828125, "learning_rate": 1e-06, "loss": 0.1766, "num_tokens": 12446741.0, "reward": -6.08203125, "reward_std": 6.159107685089111, "rewards/rm_reward_func/mean": -6.08203125, "rewards/rm_reward_func/std": 11.603202819824219, "step": 946 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 298.78125, "completions/mean_terminated_length": 276.7241516113281, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.7576, "grad_norm": 6.926180362701416, "kl": 0.4462890625, "learning_rate": 1e-06, "loss": 0.0145, "num_tokens": 12458686.0, "reward": 6.00732421875, "reward_std": 3.4000492095947266, "rewards/rm_reward_func/mean": 6.00732421875, "rewards/rm_reward_func/std": 8.406200408935547, "step": 947 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 415.0, "completions/mean_length": 269.1875, "completions/mean_terminated_length": 234.50001525878906, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "epoch": 0.7584, "grad_norm": 7.784516334533691, "kl": 2.767578125, "learning_rate": 1e-06, "loss": -0.0341, "num_tokens": 12469732.0, "reward": -5.19873046875, "reward_std": 12.44931697845459, "rewards/rm_reward_func/mean": -5.19873046875, "rewards/rm_reward_func/std": 13.344382286071777, "step": 948 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 468.0, "completions/mean_length": 263.5, "completions/mean_terminated_length": 246.933349609375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "epoch": 0.7592, "grad_norm": 7.466695785522461, "kl": 1.8359375, "learning_rate": 1e-06, "loss": 0.0924, "num_tokens": 12483748.0, "reward": -2.9599609375, "reward_std": 8.385852813720703, "rewards/rm_reward_func/mean": -2.9599609375, "rewards/rm_reward_func/std": 15.177970886230469, "step": 949 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 242.59375, "completions/mean_terminated_length": 233.90321350097656, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.76, "grad_norm": 14.014596939086914, "kl": 1.61474609375, "learning_rate": 1e-06, "loss": -0.0958, "num_tokens": 12494015.0, "reward": 1.893310546875, "reward_std": 7.712099075317383, "rewards/rm_reward_func/mean": 1.893310546875, "rewards/rm_reward_func/std": 14.392072677612305, "step": 950 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 370.1875, "completions/mean_terminated_length": 314.6956481933594, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "epoch": 0.7608, "grad_norm": 9.24715805053711, "kl": 1.931640625, "learning_rate": 1e-06, "loss": 0.0885, "num_tokens": 12508949.0, "reward": -0.7783203125, "reward_std": 8.970268249511719, "rewards/rm_reward_func/mean": -0.7783203125, "rewards/rm_reward_func/std": 15.02639102935791, "step": 951 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 474.0, "completions/mean_length": 297.5625, "completions/mean_terminated_length": 290.6451416015625, "completions/min_length": 16.0, "completions/min_terminated_length": 16.0, "epoch": 0.7616, "grad_norm": 13.282108306884766, "kl": 2.541015625, "learning_rate": 1e-06, "loss": 0.0898, "num_tokens": 12522735.0, "reward": 3.421630859375, "reward_std": 8.37659740447998, "rewards/rm_reward_func/mean": 3.421630859375, "rewards/rm_reward_func/std": 12.819450378417969, "step": 952 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 493.0, "completions/mean_length": 328.65625, "completions/mean_terminated_length": 322.7419128417969, "completions/min_length": 101.0, "completions/min_terminated_length": 101.0, "epoch": 0.7624, "grad_norm": 9.377727508544922, "kl": 1.41259765625, "learning_rate": 1e-06, "loss": -0.0102, "num_tokens": 12535252.0, "reward": 4.1376953125, "reward_std": 6.576689720153809, "rewards/rm_reward_func/mean": 4.1376953125, "rewards/rm_reward_func/std": 16.938846588134766, "step": 953 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 474.0, "completions/mean_length": 379.1875, "completions/mean_terminated_length": 360.21429443359375, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "epoch": 0.7632, "grad_norm": 13.483288764953613, "kl": 1.65625, "learning_rate": 1e-06, "loss": -0.0113, "num_tokens": 12549658.0, "reward": 4.1842041015625, "reward_std": 11.937593460083008, "rewards/rm_reward_func/mean": 4.1842041015625, "rewards/rm_reward_func/std": 16.32895851135254, "step": 954 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 351.65625, "completions/mean_terminated_length": 340.9666748046875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "epoch": 0.764, "grad_norm": 7.8393707275390625, "kl": 2.38818359375, "learning_rate": 1e-06, "loss": 0.0489, "num_tokens": 12563167.0, "reward": -6.7680511474609375, "reward_std": 6.109908103942871, "rewards/rm_reward_func/mean": -6.7680511474609375, "rewards/rm_reward_func/std": 9.325031280517578, "step": 955 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 507.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 253.34375, "completions/mean_terminated_length": 253.34375, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "epoch": 0.7648, "grad_norm": 6.62860631942749, "kl": 0.927734375, "learning_rate": 1e-06, "loss": -0.0399, "num_tokens": 12576426.0, "reward": -0.5888671875, "reward_std": 4.022240161895752, "rewards/rm_reward_func/mean": -0.5888671875, "rewards/rm_reward_func/std": 6.941095352172852, "step": 956 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 491.0, "completions/mean_length": 290.53125, "completions/mean_terminated_length": 267.6206970214844, "completions/min_length": 110.0, "completions/min_terminated_length": 110.0, "epoch": 0.7656, "grad_norm": 9.683281898498535, "kl": 4.0126953125, "learning_rate": 1e-06, "loss": 0.0822, "num_tokens": 12587875.0, "reward": -11.738689422607422, "reward_std": 5.651496887207031, "rewards/rm_reward_func/mean": -11.738689422607422, "rewards/rm_reward_func/std": 10.464712142944336, "step": 957 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 452.0, "completions/mean_length": 278.40625, "completions/mean_terminated_length": 254.2413787841797, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "epoch": 0.7664, "grad_norm": 13.09759521484375, "kl": 2.10693359375, "learning_rate": 1e-06, "loss": 0.0637, "num_tokens": 12599120.0, "reward": -3.7507591247558594, "reward_std": 5.601696014404297, "rewards/rm_reward_func/mean": -3.7507591247558594, "rewards/rm_reward_func/std": 7.122979640960693, "step": 958 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 385.4375, "completions/mean_terminated_length": 362.0, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "epoch": 0.7672, "grad_norm": 17.111623764038086, "kl": 2.60302734375, "learning_rate": 1e-06, "loss": 0.0545, "num_tokens": 12614662.0, "reward": 11.8017578125, "reward_std": 12.20899772644043, "rewards/rm_reward_func/mean": 11.8017578125, "rewards/rm_reward_func/std": 16.13597297668457, "step": 959 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 476.0, "completions/mean_length": 285.59375, "completions/mean_terminated_length": 270.5, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "epoch": 0.768, "grad_norm": 18.32476806640625, "kl": 2.861328125, "learning_rate": 1e-06, "loss": 0.1135, "num_tokens": 12628177.0, "reward": 10.161865234375, "reward_std": 9.773149490356445, "rewards/rm_reward_func/mean": 10.161865234375, "rewards/rm_reward_func/std": 18.04802131652832, "step": 960 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 314.96875, "completions/mean_terminated_length": 294.5862121582031, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "epoch": 0.7688, "grad_norm": 11.534658432006836, "kl": 2.671875, "learning_rate": 1e-06, "loss": 0.1081, "num_tokens": 12640928.0, "reward": -3.29248046875, "reward_std": 11.981504440307617, "rewards/rm_reward_func/mean": -3.29248046875, "rewards/rm_reward_func/std": 14.41004753112793, "step": 961 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 430.0, "completions/max_terminated_length": 430.0, "completions/mean_length": 194.5, "completions/mean_terminated_length": 194.5, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "epoch": 0.7696, "grad_norm": 6.043729782104492, "kl": 0.4736328125, "learning_rate": 1e-06, "loss": -0.0156, "num_tokens": 12653056.0, "reward": -7.85498046875, "reward_std": 1.4552382230758667, "rewards/rm_reward_func/mean": -7.85498046875, "rewards/rm_reward_func/std": 5.594080924987793, "step": 962 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 284.84375, "completions/mean_terminated_length": 269.70001220703125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "epoch": 0.7704, "grad_norm": 9.446344375610352, "kl": 1.1533203125, "learning_rate": 1e-06, "loss": 0.0328, "num_tokens": 12664139.0, "reward": 1.72314453125, "reward_std": 7.9739813804626465, "rewards/rm_reward_func/mean": 1.72314453125, "rewards/rm_reward_func/std": 14.244680404663086, "step": 963 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 287.34375, "completions/mean_terminated_length": 280.0967712402344, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.7712, "grad_norm": 9.243260383605957, "kl": 0.39892578125, "learning_rate": 1e-06, "loss": -0.0251, "num_tokens": 12675110.0, "reward": 13.383621215820312, "reward_std": 4.858243942260742, "rewards/rm_reward_func/mean": 13.383621215820312, "rewards/rm_reward_func/std": 12.968853950500488, "step": 964 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 404.0, "completions/mean_length": 262.125, "completions/mean_terminated_length": 192.1599884033203, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "epoch": 0.772, "grad_norm": 13.435283660888672, "kl": 4.7197265625, "learning_rate": 1e-06, "loss": 0.2661, "num_tokens": 12689778.0, "reward": -4.897491455078125, "reward_std": 4.347750663757324, "rewards/rm_reward_func/mean": -4.897491455078125, "rewards/rm_reward_func/std": 11.386714935302734, "step": 965 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 386.4375, "completions/mean_terminated_length": 344.5833435058594, "completions/min_length": 201.0, "completions/min_terminated_length": 201.0, "epoch": 0.7728, "grad_norm": 6.507021903991699, "kl": 1.4296875, "learning_rate": 1e-06, "loss": -0.0111, "num_tokens": 12705064.0, "reward": 13.21240234375, "reward_std": 10.414816856384277, "rewards/rm_reward_func/mean": 13.21240234375, "rewards/rm_reward_func/std": 14.156932830810547, "step": 966 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 473.0, "completions/max_terminated_length": 473.0, "completions/mean_length": 306.90625, "completions/mean_terminated_length": 306.90625, "completions/min_length": 108.0, "completions/min_terminated_length": 108.0, "epoch": 0.7736, "grad_norm": 6.9047393798828125, "kl": 2.828125, "learning_rate": 1e-06, "loss": -0.0205, "num_tokens": 12716701.0, "reward": 2.09149169921875, "reward_std": 10.416152954101562, "rewards/rm_reward_func/mean": 2.09149169921875, "rewards/rm_reward_func/std": 12.270896911621094, "step": 967 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 289.03125, "completions/mean_terminated_length": 281.8387145996094, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "epoch": 0.7744, "grad_norm": 7.523399353027344, "kl": 2.68798828125, "learning_rate": 1e-06, "loss": 0.0547, "num_tokens": 12729430.0, "reward": 2.884521484375, "reward_std": 5.321774482727051, "rewards/rm_reward_func/mean": 2.884521484375, "rewards/rm_reward_func/std": 15.252958297729492, "step": 968 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 423.0, "completions/mean_length": 210.21875, "completions/mean_terminated_length": 200.48387145996094, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "epoch": 0.7752, "grad_norm": 30.35150909423828, "kl": 11.3837890625, "learning_rate": 1e-06, "loss": 0.6159, "num_tokens": 12740117.0, "reward": -4.417724609375, "reward_std": 6.602893352508545, "rewards/rm_reward_func/mean": -4.417724609375, "rewards/rm_reward_func/std": 15.255743980407715, "step": 969 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 353.03125, "completions/mean_terminated_length": 336.5862121582031, "completions/min_length": 107.0, "completions/min_terminated_length": 107.0, "epoch": 0.776, "grad_norm": 8.360908508300781, "kl": 2.9921875, "learning_rate": 1e-06, "loss": 0.1711, "num_tokens": 12753542.0, "reward": -0.7406158447265625, "reward_std": 4.496933937072754, "rewards/rm_reward_func/mean": -0.7406158447265625, "rewards/rm_reward_func/std": 11.36292839050293, "step": 970 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 360.15625, "completions/mean_terminated_length": 291.1363830566406, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "epoch": 0.7768, "grad_norm": 10.487771987915039, "kl": 0.97509765625, "learning_rate": 1e-06, "loss": 0.0275, "num_tokens": 12767451.0, "reward": 5.02734375, "reward_std": 7.548166275024414, "rewards/rm_reward_func/mean": 5.02734375, "rewards/rm_reward_func/std": 21.371307373046875, "step": 971 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 348.0625, "completions/mean_terminated_length": 331.10345458984375, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "epoch": 0.7776, "grad_norm": 6.754936695098877, "kl": 2.9931640625, "learning_rate": 1e-06, "loss": 0.0756, "num_tokens": 12780813.0, "reward": 0.3992156982421875, "reward_std": 5.77485466003418, "rewards/rm_reward_func/mean": 0.3992156982421875, "rewards/rm_reward_func/std": 6.508362770080566, "step": 972 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 477.0, "completions/mean_length": 350.78125, "completions/mean_terminated_length": 345.58062744140625, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "epoch": 0.7784, "grad_norm": 6.786907196044922, "kl": 1.45751953125, "learning_rate": 1e-06, "loss": -0.0457, "num_tokens": 12797526.0, "reward": 14.609710693359375, "reward_std": 11.176668167114258, "rewards/rm_reward_func/mean": 14.609710693359375, "rewards/rm_reward_func/std": 18.939729690551758, "step": 973 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 307.34375, "completions/mean_terminated_length": 250.0399932861328, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.7792, "grad_norm": 9.214887619018555, "kl": 2.23046875, "learning_rate": 1e-06, "loss": 0.0533, "num_tokens": 12811017.0, "reward": -0.3729248046875, "reward_std": 4.81978702545166, "rewards/rm_reward_func/mean": -0.3729248046875, "rewards/rm_reward_func/std": 6.530143737792969, "step": 974 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 476.0, "completions/mean_length": 307.28125, "completions/mean_terminated_length": 293.63336181640625, "completions/min_length": 112.0, "completions/min_terminated_length": 112.0, "epoch": 0.78, "grad_norm": 28.01193618774414, "kl": 11.6064453125, "learning_rate": 1e-06, "loss": 0.4855, "num_tokens": 12823162.0, "reward": -6.5583343505859375, "reward_std": 6.375615119934082, "rewards/rm_reward_func/mean": -6.5583343505859375, "rewards/rm_reward_func/std": 13.018743515014648, "step": 975 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 359.5, "completions/mean_terminated_length": 331.2592468261719, "completions/min_length": 148.0, "completions/min_terminated_length": 148.0, "epoch": 0.7808, "grad_norm": 16.22539710998535, "kl": 6.41015625, "learning_rate": 1e-06, "loss": 0.1488, "num_tokens": 12837546.0, "reward": -6.0869140625, "reward_std": 5.604630470275879, "rewards/rm_reward_func/mean": -6.0869140625, "rewards/rm_reward_func/std": 11.859865188598633, "step": 976 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 485.0, "completions/mean_length": 297.25, "completions/mean_terminated_length": 237.1199951171875, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "epoch": 0.7816, "grad_norm": 43.490577697753906, "kl": 7.90283203125, "learning_rate": 1e-06, "loss": 0.1253, "num_tokens": 12850946.0, "reward": -4.4287109375, "reward_std": 7.8558669090271, "rewards/rm_reward_func/mean": -4.4287109375, "rewards/rm_reward_func/std": 18.278430938720703, "step": 977 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 333.0, "completions/max_terminated_length": 333.0, "completions/mean_length": 167.125, "completions/mean_terminated_length": 167.125, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "epoch": 0.7824, "grad_norm": 9.18833065032959, "kl": 2.99072265625, "learning_rate": 1e-06, "loss": 0.2724, "num_tokens": 12858686.0, "reward": 2.391876220703125, "reward_std": 3.7164864540100098, "rewards/rm_reward_func/mean": 2.391876220703125, "rewards/rm_reward_func/std": 9.332040786743164, "step": 978 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 340.0625, "completions/mean_terminated_length": 334.51611328125, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "epoch": 0.7832, "grad_norm": 6.65054988861084, "kl": 0.5810546875, "learning_rate": 1e-06, "loss": -0.0442, "num_tokens": 12872840.0, "reward": 8.1917724609375, "reward_std": 7.095458984375, "rewards/rm_reward_func/mean": 8.1917724609375, "rewards/rm_reward_func/std": 12.081109046936035, "step": 979 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 403.0, "completions/mean_terminated_length": 382.8148193359375, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "epoch": 0.784, "grad_norm": 9.431859970092773, "kl": 7.0546875, "learning_rate": 1e-06, "loss": 0.2173, "num_tokens": 12888208.0, "reward": 2.475616455078125, "reward_std": 8.312936782836914, "rewards/rm_reward_func/mean": 2.475616455078125, "rewards/rm_reward_func/std": 13.530970573425293, "step": 980 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 407.84375, "completions/mean_terminated_length": 383.8077087402344, "completions/min_length": 82.0, "completions/min_terminated_length": 82.0, "epoch": 0.7848, "grad_norm": 6.22176456451416, "kl": 3.5087890625, "learning_rate": 1e-06, "loss": 0.0194, "num_tokens": 12907155.0, "reward": 4.936767578125, "reward_std": 9.09925651550293, "rewards/rm_reward_func/mean": 4.936767578125, "rewards/rm_reward_func/std": 20.013263702392578, "step": 981 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 320.96875, "completions/mean_terminated_length": 301.2069091796875, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "epoch": 0.7856, "grad_norm": 5.159428119659424, "kl": 1.8134765625, "learning_rate": 1e-06, "loss": 0.0296, "num_tokens": 12921042.0, "reward": 9.160308837890625, "reward_std": 9.23448371887207, "rewards/rm_reward_func/mean": 9.160308837890625, "rewards/rm_reward_func/std": 13.724321365356445, "step": 982 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 426.0, "completions/max_terminated_length": 426.0, "completions/mean_length": 257.75, "completions/mean_terminated_length": 257.75, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "epoch": 0.7864, "grad_norm": 10.37830924987793, "kl": 1.292724609375, "learning_rate": 1e-06, "loss": -0.0107, "num_tokens": 12932818.0, "reward": 3.912353515625, "reward_std": 6.3594069480896, "rewards/rm_reward_func/mean": 3.912353515625, "rewards/rm_reward_func/std": 14.007357597351074, "step": 983 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.53125, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 390.03125, "completions/mean_terminated_length": 251.80001831054688, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "epoch": 0.7872, "grad_norm": 8.068547248840332, "kl": 0.436279296875, "learning_rate": 1e-06, "loss": 0.0255, "num_tokens": 12951635.0, "reward": 4.229248046875, "reward_std": 4.562774658203125, "rewards/rm_reward_func/mean": 4.229248046875, "rewards/rm_reward_func/std": 10.442963600158691, "step": 984 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 262.125, "completions/mean_terminated_length": 204.4615478515625, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "epoch": 0.788, "grad_norm": 11.331951141357422, "kl": 0.728759765625, "learning_rate": 1e-06, "loss": 0.0525, "num_tokens": 12965559.0, "reward": 3.13623046875, "reward_std": 7.865469932556152, "rewards/rm_reward_func/mean": 3.13623046875, "rewards/rm_reward_func/std": 14.55949592590332, "step": 985 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 462.0, "completions/mean_length": 294.71875, "completions/mean_terminated_length": 287.70965576171875, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "epoch": 0.7888, "grad_norm": 12.841826438903809, "kl": 0.88623046875, "learning_rate": 1e-06, "loss": 0.1087, "num_tokens": 12977742.0, "reward": 5.2115478515625, "reward_std": 3.0502023696899414, "rewards/rm_reward_func/mean": 5.2115478515625, "rewards/rm_reward_func/std": 14.679587364196777, "step": 986 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 475.0, "completions/max_terminated_length": 475.0, "completions/mean_length": 222.34375, "completions/mean_terminated_length": 222.34375, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "epoch": 0.7896, "grad_norm": 6.139382839202881, "kl": 1.00048828125, "learning_rate": 1e-06, "loss": 0.0219, "num_tokens": 12987129.0, "reward": 1.97021484375, "reward_std": 4.622243881225586, "rewards/rm_reward_func/mean": 1.97021484375, "rewards/rm_reward_func/std": 9.068449020385742, "step": 987 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 506.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 268.15625, "completions/mean_terminated_length": 268.15625, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "epoch": 0.7904, "grad_norm": 9.928857803344727, "kl": 1.89697265625, "learning_rate": 1e-06, "loss": -0.1183, "num_tokens": 12999926.0, "reward": -0.8626174926757812, "reward_std": 6.4827094078063965, "rewards/rm_reward_func/mean": -0.8626174926757812, "rewards/rm_reward_func/std": 8.545820236206055, "step": 988 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 446.03125, "completions/mean_terminated_length": 436.6071472167969, "completions/min_length": 260.0, "completions/min_terminated_length": 260.0, "epoch": 0.7912, "grad_norm": 8.559710502624512, "kl": 2.8173828125, "learning_rate": 1e-06, "loss": 0.1298, "num_tokens": 13017151.0, "reward": -2.3729248046875, "reward_std": 6.979788303375244, "rewards/rm_reward_func/mean": -2.3729248046875, "rewards/rm_reward_func/std": 10.80506706237793, "step": 989 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 488.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 360.53125, "completions/mean_terminated_length": 360.53125, "completions/min_length": 228.0, "completions/min_terminated_length": 228.0, "epoch": 0.792, "grad_norm": 14.2783784866333, "kl": 4.57421875, "learning_rate": 1e-06, "loss": 0.2056, "num_tokens": 13031040.0, "reward": 3.993865966796875, "reward_std": 10.918571472167969, "rewards/rm_reward_func/mean": 3.993865966796875, "rewards/rm_reward_func/std": 11.101519584655762, "step": 990 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 418.0, "completions/mean_length": 253.21875, "completions/mean_terminated_length": 205.29629516601562, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "epoch": 0.7928, "grad_norm": 12.58895492553711, "kl": 4.3525390625, "learning_rate": 1e-06, "loss": 0.1975, "num_tokens": 13041079.0, "reward": -6.41998291015625, "reward_std": 8.907421112060547, "rewards/rm_reward_func/mean": -6.41998291015625, "rewards/rm_reward_func/std": 10.190556526184082, "step": 991 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 399.0, "completions/max_terminated_length": 399.0, "completions/mean_length": 209.8125, "completions/mean_terminated_length": 209.8125, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.7936, "grad_norm": 8.097042083740234, "kl": 4.4404296875, "learning_rate": 1e-06, "loss": 0.1805, "num_tokens": 13053145.0, "reward": 5.662109375, "reward_std": 10.280261993408203, "rewards/rm_reward_func/mean": 5.662109375, "rewards/rm_reward_func/std": 12.116209030151367, "step": 992 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 193.875, "completions/mean_terminated_length": 172.6666717529297, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.7944, "grad_norm": 16.431880950927734, "kl": 3.3447265625, "learning_rate": 1e-06, "loss": 0.1053, "num_tokens": 13064117.0, "reward": 2.54443359375, "reward_std": 3.539630889892578, "rewards/rm_reward_func/mean": 2.54443359375, "rewards/rm_reward_func/std": 13.015660285949707, "step": 993 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 325.3125, "completions/mean_terminated_length": 240.45455932617188, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "epoch": 0.7952, "grad_norm": 5.3214592933654785, "kl": 2.08642578125, "learning_rate": 1e-06, "loss": 0.1091, "num_tokens": 13080759.0, "reward": 5.818359375, "reward_std": 6.74766731262207, "rewards/rm_reward_func/mean": 5.818359375, "rewards/rm_reward_func/std": 13.01656723022461, "step": 994 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 440.0, "completions/mean_length": 398.0625, "completions/mean_terminated_length": 360.0833435058594, "completions/min_length": 114.0, "completions/min_terminated_length": 114.0, "epoch": 0.796, "grad_norm": 7.939972400665283, "kl": 3.1787109375, "learning_rate": 1e-06, "loss": 0.0815, "num_tokens": 13096265.0, "reward": -3.901641845703125, "reward_std": 7.492610931396484, "rewards/rm_reward_func/mean": -3.901641845703125, "rewards/rm_reward_func/std": 12.713387489318848, "step": 995 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 433.0, "completions/max_terminated_length": 433.0, "completions/mean_length": 216.65625, "completions/mean_terminated_length": 216.65625, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.7968, "grad_norm": 17.098487854003906, "kl": 3.54638671875, "learning_rate": 1e-06, "loss": 0.0526, "num_tokens": 13109534.0, "reward": -1.665863037109375, "reward_std": 4.5015339851379395, "rewards/rm_reward_func/mean": -1.665863037109375, "rewards/rm_reward_func/std": 8.465290069580078, "step": 996 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 354.09375, "completions/mean_terminated_length": 317.65386962890625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "epoch": 0.7976, "grad_norm": 9.26326847076416, "kl": 0.84375, "learning_rate": 1e-06, "loss": 0.0522, "num_tokens": 13124065.0, "reward": 9.808670043945312, "reward_std": 6.167551040649414, "rewards/rm_reward_func/mean": 9.808670043945312, "rewards/rm_reward_func/std": 7.601739406585693, "step": 997 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 466.0, "completions/mean_length": 373.6875, "completions/mean_terminated_length": 364.4666748046875, "completions/min_length": 240.0, "completions/min_terminated_length": 240.0, "epoch": 0.7984, "grad_norm": 9.506736755371094, "kl": 7.11279296875, "learning_rate": 1e-06, "loss": 0.3447, "num_tokens": 13138111.0, "reward": -5.72540283203125, "reward_std": 7.122351169586182, "rewards/rm_reward_func/mean": -5.72540283203125, "rewards/rm_reward_func/std": 7.666444301605225, "step": 998 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 406.0, "completions/mean_length": 242.625, "completions/mean_terminated_length": 167.1999969482422, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "epoch": 0.7992, "grad_norm": 11.145411491394043, "kl": 1.321533203125, "learning_rate": 1e-06, "loss": 0.0769, "num_tokens": 13149155.0, "reward": 4.50341796875, "reward_std": 3.2994678020477295, "rewards/rm_reward_func/mean": 4.50341796875, "rewards/rm_reward_func/std": 8.315497398376465, "step": 999 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.59375, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 479.65625, "completions/mean_terminated_length": 432.3846435546875, "completions/min_length": 278.0, "completions/min_terminated_length": 278.0, "epoch": 0.8, "grad_norm": 10.525664329528809, "kl": 3.92578125, "learning_rate": 1e-06, "loss": 0.121, "num_tokens": 13169528.0, "reward": -4.1995849609375, "reward_std": 10.804488182067871, "rewards/rm_reward_func/mean": -4.1995849609375, "rewards/rm_reward_func/std": 12.529898643493652, "step": 1000 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 345.59375, "completions/mean_terminated_length": 269.9545593261719, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "epoch": 0.8008, "grad_norm": 8.35184383392334, "kl": 2.693359375, "learning_rate": 1e-06, "loss": 0.0605, "num_tokens": 13182515.0, "reward": 0.7723388671875, "reward_std": 5.575720310211182, "rewards/rm_reward_func/mean": 0.7723388671875, "rewards/rm_reward_func/std": 9.116813659667969, "step": 1001 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 476.0, "completions/mean_length": 395.625, "completions/mean_terminated_length": 379.0000305175781, "completions/min_length": 204.0, "completions/min_terminated_length": 204.0, "epoch": 0.8016, "grad_norm": 9.205164909362793, "kl": 3.8544921875, "learning_rate": 1e-06, "loss": 0.1071, "num_tokens": 13197319.0, "reward": 12.017333984375, "reward_std": 12.300132751464844, "rewards/rm_reward_func/mean": 12.017333984375, "rewards/rm_reward_func/std": 18.185022354125977, "step": 1002 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 391.875, "completions/mean_terminated_length": 344.86956787109375, "completions/min_length": 148.0, "completions/min_terminated_length": 148.0, "epoch": 0.8024, "grad_norm": 7.878791332244873, "kl": 2.18896484375, "learning_rate": 1e-06, "loss": 0.0596, "num_tokens": 13212147.0, "reward": -5.649444580078125, "reward_std": 5.377230167388916, "rewards/rm_reward_func/mean": -5.649444580078125, "rewards/rm_reward_func/std": 7.007201671600342, "step": 1003 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 304.0, "completions/max_terminated_length": 304.0, "completions/mean_length": 119.34375, "completions/mean_terminated_length": 119.34375, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "epoch": 0.8032, "grad_norm": 6.185940742492676, "kl": 1.81640625, "learning_rate": 1e-06, "loss": 0.068, "num_tokens": 13220862.0, "reward": 4.4765625, "reward_std": 0.8225632309913635, "rewards/rm_reward_func/mean": 4.4765625, "rewards/rm_reward_func/std": 10.800799369812012, "step": 1004 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 359.25, "completions/mean_terminated_length": 299.478271484375, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.804, "grad_norm": 38.11021041870117, "kl": 4.56787109375, "learning_rate": 1e-06, "loss": 0.0942, "num_tokens": 13235742.0, "reward": -5.1068115234375, "reward_std": 9.743478775024414, "rewards/rm_reward_func/mean": -5.1068115234375, "rewards/rm_reward_func/std": 13.0623197555542, "step": 1005 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 473.0, "completions/mean_length": 213.75, "completions/mean_terminated_length": 204.1290283203125, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "epoch": 0.8048, "grad_norm": 13.769661903381348, "kl": 1.0166015625, "learning_rate": 1e-06, "loss": -0.0556, "num_tokens": 13245302.0, "reward": -6.537109375, "reward_std": 4.325830459594727, "rewards/rm_reward_func/mean": -6.537109375, "rewards/rm_reward_func/std": 9.237016677856445, "step": 1006 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 281.875, "completions/mean_terminated_length": 266.5333557128906, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "epoch": 0.8056, "grad_norm": 13.841572761535645, "kl": 5.92578125, "learning_rate": 1e-06, "loss": 0.1448, "num_tokens": 13256546.0, "reward": -1.900146484375, "reward_std": 8.712725639343262, "rewards/rm_reward_func/mean": -1.900146484375, "rewards/rm_reward_func/std": 12.52529525756836, "step": 1007 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 501.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 309.78125, "completions/mean_terminated_length": 309.78125, "completions/min_length": 23.0, "completions/min_terminated_length": 23.0, "epoch": 0.8064, "grad_norm": 21.372207641601562, "kl": 4.873291015625, "learning_rate": 1e-06, "loss": 0.0049, "num_tokens": 13268731.0, "reward": 5.0888671875, "reward_std": 5.007624626159668, "rewards/rm_reward_func/mean": 5.0888671875, "rewards/rm_reward_func/std": 22.30028533935547, "step": 1008 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 314.96875, "completions/mean_terminated_length": 286.8214416503906, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "epoch": 0.8072, "grad_norm": 25.30167579650879, "kl": 9.71826171875, "learning_rate": 1e-06, "loss": 0.4105, "num_tokens": 13285794.0, "reward": -8.27099609375, "reward_std": 7.645841598510742, "rewards/rm_reward_func/mean": -8.27099609375, "rewards/rm_reward_func/std": 16.137731552124023, "step": 1009 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 410.0, "completions/mean_length": 273.875, "completions/mean_terminated_length": 266.19354248046875, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "epoch": 0.808, "grad_norm": 8.614705085754395, "kl": 0.37060546875, "learning_rate": 1e-06, "loss": -0.0065, "num_tokens": 13299302.0, "reward": 20.390045166015625, "reward_std": 5.194117546081543, "rewards/rm_reward_func/mean": 20.390045166015625, "rewards/rm_reward_func/std": 17.19424057006836, "step": 1010 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 342.6875, "completions/mean_terminated_length": 303.6153869628906, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.8088, "grad_norm": 4.921807289123535, "kl": 1.82275390625, "learning_rate": 1e-06, "loss": 0.0481, "num_tokens": 13313140.0, "reward": 4.9627838134765625, "reward_std": 6.066807270050049, "rewards/rm_reward_func/mean": 4.9627838134765625, "rewards/rm_reward_func/std": 9.324835777282715, "step": 1011 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 455.0, "completions/mean_length": 245.84375, "completions/mean_terminated_length": 171.3199920654297, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "epoch": 0.8096, "grad_norm": 19.971662521362305, "kl": 3.2412109375, "learning_rate": 1e-06, "loss": 0.0408, "num_tokens": 13325503.0, "reward": -3.34765625, "reward_std": 4.555385589599609, "rewards/rm_reward_func/mean": -3.34765625, "rewards/rm_reward_func/std": 15.34835433959961, "step": 1012 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 365.0, "completions/max_terminated_length": 365.0, "completions/mean_length": 193.03125, "completions/mean_terminated_length": 193.03125, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "epoch": 0.8104, "grad_norm": 11.546648025512695, "kl": 1.3896484375, "learning_rate": 1e-06, "loss": 0.0518, "num_tokens": 13334680.0, "reward": -4.6224365234375, "reward_std": 3.3871943950653076, "rewards/rm_reward_func/mean": -4.6224365234375, "rewards/rm_reward_func/std": 9.441699981689453, "step": 1013 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 431.0, "completions/max_terminated_length": 431.0, "completions/mean_length": 322.84375, "completions/mean_terminated_length": 322.84375, "completions/min_length": 155.0, "completions/min_terminated_length": 155.0, "epoch": 0.8112, "grad_norm": 6.9250359535217285, "kl": 1.911376953125, "learning_rate": 1e-06, "loss": 0.0922, "num_tokens": 13347131.0, "reward": 4.8478851318359375, "reward_std": 8.108898162841797, "rewards/rm_reward_func/mean": 4.8478851318359375, "rewards/rm_reward_func/std": 15.027045249938965, "step": 1014 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 232.625, "completions/mean_terminated_length": 192.71429443359375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "epoch": 0.812, "grad_norm": 35.626102447509766, "kl": 0.970703125, "learning_rate": 1e-06, "loss": 0.0813, "num_tokens": 13360111.0, "reward": 0.5765380859375, "reward_std": 6.022714614868164, "rewards/rm_reward_func/mean": 0.5765380859375, "rewards/rm_reward_func/std": 14.149896621704102, "step": 1015 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 380.0, "completions/mean_length": 352.46875, "completions/mean_terminated_length": 299.29168701171875, "completions/min_length": 220.0, "completions/min_terminated_length": 220.0, "epoch": 0.8128, "grad_norm": 9.04706859588623, "kl": 0.98681640625, "learning_rate": 1e-06, "loss": 0.0161, "num_tokens": 13373950.0, "reward": -5.08154296875, "reward_std": 3.3396553993225098, "rewards/rm_reward_func/mean": -5.08154296875, "rewards/rm_reward_func/std": 7.902137756347656, "step": 1016 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 414.5, "completions/mean_terminated_length": 392.0, "completions/min_length": 206.0, "completions/min_terminated_length": 206.0, "epoch": 0.8136, "grad_norm": 7.317805290222168, "kl": 0.775390625, "learning_rate": 1e-06, "loss": -0.0302, "num_tokens": 13391822.0, "reward": 5.2294158935546875, "reward_std": 6.866477012634277, "rewards/rm_reward_func/mean": 5.2294158935546875, "rewards/rm_reward_func/std": 10.840312957763672, "step": 1017 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 267.6875, "completions/mean_terminated_length": 222.44444274902344, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "epoch": 0.8144, "grad_norm": 8.47047233581543, "kl": 0.93603515625, "learning_rate": 1e-06, "loss": 0.0759, "num_tokens": 13402404.0, "reward": 0.318359375, "reward_std": 2.553496837615967, "rewards/rm_reward_func/mean": 0.318359375, "rewards/rm_reward_func/std": 9.620682716369629, "step": 1018 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 509.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 298.75, "completions/mean_terminated_length": 298.75, "completions/min_length": 91.0, "completions/min_terminated_length": 91.0, "epoch": 0.8152, "grad_norm": 10.068121910095215, "kl": 0.60400390625, "learning_rate": 1e-06, "loss": 0.0192, "num_tokens": 13414804.0, "reward": -0.441314697265625, "reward_std": 2.4171152114868164, "rewards/rm_reward_func/mean": -0.441314697265625, "rewards/rm_reward_func/std": 6.881007671356201, "step": 1019 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 498.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 343.5, "completions/mean_terminated_length": 343.5, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "epoch": 0.816, "grad_norm": 10.040746688842773, "kl": 3.17138671875, "learning_rate": 1e-06, "loss": 0.005, "num_tokens": 13428028.0, "reward": 8.162225723266602, "reward_std": 8.07081413269043, "rewards/rm_reward_func/mean": 8.162225723266602, "rewards/rm_reward_func/std": 20.003002166748047, "step": 1020 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 482.0, "completions/mean_length": 391.78125, "completions/mean_terminated_length": 351.7083435058594, "completions/min_length": 187.0, "completions/min_terminated_length": 187.0, "epoch": 0.8168, "grad_norm": 10.024792671203613, "kl": 2.970703125, "learning_rate": 1e-06, "loss": 0.0447, "num_tokens": 13443005.0, "reward": -0.19919586181640625, "reward_std": 7.3775715827941895, "rewards/rm_reward_func/mean": -0.19919586181640625, "rewards/rm_reward_func/std": 14.476346015930176, "step": 1021 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 340.09375, "completions/mean_terminated_length": 334.5483703613281, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "epoch": 0.8176, "grad_norm": 7.288039684295654, "kl": 0.75927734375, "learning_rate": 1e-06, "loss": 0.001, "num_tokens": 13456536.0, "reward": 9.702880859375, "reward_std": 5.229021072387695, "rewards/rm_reward_func/mean": 9.702880859375, "rewards/rm_reward_func/std": 9.884917259216309, "step": 1022 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 433.0, "completions/max_terminated_length": 433.0, "completions/mean_length": 216.71875, "completions/mean_terminated_length": 216.71875, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "epoch": 0.8184, "grad_norm": 5.475179672241211, "kl": 0.51953125, "learning_rate": 1e-06, "loss": 0.0019, "num_tokens": 13466887.0, "reward": 9.95068359375, "reward_std": 3.0980048179626465, "rewards/rm_reward_func/mean": 9.95068359375, "rewards/rm_reward_func/std": 5.0042338371276855, "step": 1023 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 334.65625, "completions/mean_terminated_length": 301.8148193359375, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "epoch": 0.8192, "grad_norm": 9.030261039733887, "kl": 0.4716796875, "learning_rate": 1e-06, "loss": -0.0265, "num_tokens": 13480036.0, "reward": 2.835418701171875, "reward_std": 5.5337724685668945, "rewards/rm_reward_func/mean": 2.835418701171875, "rewards/rm_reward_func/std": 8.441008567810059, "step": 1024 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 472.0, "completions/mean_length": 232.0625, "completions/mean_terminated_length": 223.03225708007812, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "epoch": 0.82, "grad_norm": 13.575851440429688, "kl": 1.310546875, "learning_rate": 1e-06, "loss": 0.0396, "num_tokens": 13491534.0, "reward": 4.1461181640625, "reward_std": 4.509701728820801, "rewards/rm_reward_func/mean": 4.1461181640625, "rewards/rm_reward_func/std": 10.166535377502441, "step": 1025 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 408.0, "completions/max_terminated_length": 408.0, "completions/mean_length": 250.875, "completions/mean_terminated_length": 250.875, "completions/min_length": 95.0, "completions/min_terminated_length": 95.0, "epoch": 0.8208, "grad_norm": 10.076627731323242, "kl": 3.59375, "learning_rate": 1e-06, "loss": -0.0269, "num_tokens": 13505082.0, "reward": 2.3536376953125, "reward_std": 10.482784271240234, "rewards/rm_reward_func/mean": 2.3536376953125, "rewards/rm_reward_func/std": 24.783512115478516, "step": 1026 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 411.0, "completions/max_terminated_length": 411.0, "completions/mean_length": 260.21875, "completions/mean_terminated_length": 260.21875, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "epoch": 0.8216, "grad_norm": 27.619287490844727, "kl": 2.212890625, "learning_rate": 1e-06, "loss": 0.0159, "num_tokens": 13515673.0, "reward": 0.1591796875, "reward_std": 5.797090530395508, "rewards/rm_reward_func/mean": 0.1591796875, "rewards/rm_reward_func/std": 17.642372131347656, "step": 1027 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 446.0, "completions/mean_length": 341.03125, "completions/mean_terminated_length": 316.6071472167969, "completions/min_length": 154.0, "completions/min_terminated_length": 154.0, "epoch": 0.8224, "grad_norm": 8.583765029907227, "kl": 2.4423828125, "learning_rate": 1e-06, "loss": 0.072, "num_tokens": 13531434.0, "reward": 7.218017578125, "reward_std": 3.8263235092163086, "rewards/rm_reward_func/mean": 7.218017578125, "rewards/rm_reward_func/std": 22.30869483947754, "step": 1028 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 301.0, "completions/mean_terminated_length": 294.19354248046875, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "epoch": 0.8232, "grad_norm": 17.819374084472656, "kl": 0.42431640625, "learning_rate": 1e-06, "loss": 0.0075, "num_tokens": 13543170.0, "reward": 13.765625, "reward_std": 3.121476173400879, "rewards/rm_reward_func/mean": 13.765625, "rewards/rm_reward_func/std": 17.44659996032715, "step": 1029 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 463.0, "completions/mean_length": 335.4375, "completions/mean_terminated_length": 255.18182373046875, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "epoch": 0.824, "grad_norm": 12.179718971252441, "kl": 1.9833984375, "learning_rate": 1e-06, "loss": 0.048, "num_tokens": 13558328.0, "reward": -1.099609375, "reward_std": 2.019794464111328, "rewards/rm_reward_func/mean": -1.099609375, "rewards/rm_reward_func/std": 11.247794151306152, "step": 1030 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 479.0, "completions/mean_length": 319.6875, "completions/mean_terminated_length": 255.58334350585938, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "epoch": 0.8248, "grad_norm": 5.296995162963867, "kl": 0.63330078125, "learning_rate": 1e-06, "loss": -0.0521, "num_tokens": 13571750.0, "reward": 8.6865234375, "reward_std": 5.767685890197754, "rewards/rm_reward_func/mean": 8.6865234375, "rewards/rm_reward_func/std": 10.389171600341797, "step": 1031 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 244.0625, "completions/mean_terminated_length": 216.34483337402344, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "epoch": 0.8256, "grad_norm": 8.053586959838867, "kl": 2.3212890625, "learning_rate": 1e-06, "loss": 0.034, "num_tokens": 13581456.0, "reward": -4.644775390625, "reward_std": 6.1426496505737305, "rewards/rm_reward_func/mean": -4.644775390625, "rewards/rm_reward_func/std": 9.984515190124512, "step": 1032 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 485.0, "completions/mean_length": 285.1875, "completions/mean_terminated_length": 261.7241516113281, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "epoch": 0.8264, "grad_norm": 6.782697677612305, "kl": 0.65087890625, "learning_rate": 1e-06, "loss": 0.0033, "num_tokens": 13593006.0, "reward": 0.996185302734375, "reward_std": 2.4062700271606445, "rewards/rm_reward_func/mean": 0.996185302734375, "rewards/rm_reward_func/std": 5.100530624389648, "step": 1033 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 303.96875, "completions/mean_terminated_length": 297.258056640625, "completions/min_length": 118.0, "completions/min_terminated_length": 118.0, "epoch": 0.8272, "grad_norm": 11.949031829833984, "kl": 3.359375, "learning_rate": 1e-06, "loss": 0.0168, "num_tokens": 13604669.0, "reward": -4.284423828125, "reward_std": 13.229473114013672, "rewards/rm_reward_func/mean": -4.284423828125, "rewards/rm_reward_func/std": 17.82453727722168, "step": 1034 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 431.15625, "completions/mean_terminated_length": 404.2083435058594, "completions/min_length": 264.0, "completions/min_terminated_length": 264.0, "epoch": 0.828, "grad_norm": 6.652060031890869, "kl": 1.0, "learning_rate": 1e-06, "loss": 0.0325, "num_tokens": 13620410.0, "reward": 1.5401611328125, "reward_std": 7.661131858825684, "rewards/rm_reward_func/mean": 1.5401611328125, "rewards/rm_reward_func/std": 9.722722053527832, "step": 1035 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 262.53125, "completions/mean_terminated_length": 226.8928680419922, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "epoch": 0.8288, "grad_norm": 12.548855781555176, "kl": 3.18408203125, "learning_rate": 1e-06, "loss": 0.1386, "num_tokens": 13632595.0, "reward": 0.58056640625, "reward_std": 2.6908631324768066, "rewards/rm_reward_func/mean": 0.58056640625, "rewards/rm_reward_func/std": 9.962352752685547, "step": 1036 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 465.0, "completions/mean_length": 333.3125, "completions/mean_terminated_length": 327.5483703613281, "completions/min_length": 129.0, "completions/min_terminated_length": 129.0, "epoch": 0.8296, "grad_norm": 7.802600383758545, "kl": 3.310546875, "learning_rate": 1e-06, "loss": 0.081, "num_tokens": 13648397.0, "reward": -1.08026123046875, "reward_std": 7.778307914733887, "rewards/rm_reward_func/mean": -1.08026123046875, "rewards/rm_reward_func/std": 9.398695945739746, "step": 1037 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 468.0, "completions/max_terminated_length": 468.0, "completions/mean_length": 261.5, "completions/mean_terminated_length": 261.5, "completions/min_length": 118.0, "completions/min_terminated_length": 118.0, "epoch": 0.8304, "grad_norm": 13.804716110229492, "kl": 1.47998046875, "learning_rate": 1e-06, "loss": 0.0421, "num_tokens": 13659029.0, "reward": -1.4334716796875, "reward_std": 3.4379241466522217, "rewards/rm_reward_func/mean": -1.4334716796875, "rewards/rm_reward_func/std": 13.476388931274414, "step": 1038 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 397.59375, "completions/mean_terminated_length": 376.40740966796875, "completions/min_length": 129.0, "completions/min_terminated_length": 129.0, "epoch": 0.8312, "grad_norm": 11.404579162597656, "kl": 1.0322265625, "learning_rate": 1e-06, "loss": -0.0507, "num_tokens": 13675640.0, "reward": 4.508544921875, "reward_std": 11.643041610717773, "rewards/rm_reward_func/mean": 4.508544921875, "rewards/rm_reward_func/std": 12.394889831542969, "step": 1039 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 358.3125, "completions/mean_terminated_length": 315.2799987792969, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.832, "grad_norm": 7.375335216522217, "kl": 1.384765625, "learning_rate": 1e-06, "loss": 0.0147, "num_tokens": 13689202.0, "reward": 0.5595703125, "reward_std": 5.928117752075195, "rewards/rm_reward_func/mean": 0.5595703125, "rewards/rm_reward_func/std": 8.306135177612305, "step": 1040 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 442.28125, "completions/mean_terminated_length": 400.45001220703125, "completions/min_length": 245.0, "completions/min_terminated_length": 245.0, "epoch": 0.8328, "grad_norm": 103.06358337402344, "kl": 5.14306640625, "learning_rate": 1e-06, "loss": 0.2078, "num_tokens": 13706795.0, "reward": -7.9290771484375, "reward_std": 5.579919815063477, "rewards/rm_reward_func/mean": -7.9290771484375, "rewards/rm_reward_func/std": 9.49826717376709, "step": 1041 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 460.0, "completions/max_terminated_length": 460.0, "completions/mean_length": 267.0625, "completions/mean_terminated_length": 267.0625, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "epoch": 0.8336, "grad_norm": 13.862227439880371, "kl": 2.5947265625, "learning_rate": 1e-06, "loss": -0.0647, "num_tokens": 13717565.0, "reward": -8.9283447265625, "reward_std": 2.9016993045806885, "rewards/rm_reward_func/mean": -8.9283447265625, "rewards/rm_reward_func/std": 7.622008800506592, "step": 1042 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 388.78125, "completions/mean_terminated_length": 376.03448486328125, "completions/min_length": 245.0, "completions/min_terminated_length": 245.0, "epoch": 0.8344, "grad_norm": 10.778160095214844, "kl": 1.15380859375, "learning_rate": 1e-06, "loss": 0.0711, "num_tokens": 13735566.0, "reward": 7.612060546875, "reward_std": 7.548920631408691, "rewards/rm_reward_func/mean": 7.612060546875, "rewards/rm_reward_func/std": 13.313557624816895, "step": 1043 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 371.40625, "completions/mean_terminated_length": 366.8709716796875, "completions/min_length": 113.0, "completions/min_terminated_length": 113.0, "epoch": 0.8352, "grad_norm": 8.611907958984375, "kl": 3.005126953125, "learning_rate": 1e-06, "loss": 0.0158, "num_tokens": 13749723.0, "reward": 3.6009521484375, "reward_std": 10.527230262756348, "rewards/rm_reward_func/mean": 3.6009521484375, "rewards/rm_reward_func/std": 13.915834426879883, "step": 1044 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 301.0, "completions/mean_terminated_length": 279.17242431640625, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "epoch": 0.836, "grad_norm": 19.481334686279297, "kl": 6.16796875, "learning_rate": 1e-06, "loss": 0.1613, "num_tokens": 13765139.0, "reward": -4.6524658203125, "reward_std": 3.753931999206543, "rewards/rm_reward_func/mean": -4.6524658203125, "rewards/rm_reward_func/std": 11.036279678344727, "step": 1045 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 481.0, "completions/mean_length": 272.84375, "completions/mean_terminated_length": 228.55555725097656, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "epoch": 0.8368, "grad_norm": 18.560142517089844, "kl": 7.0703125, "learning_rate": 1e-06, "loss": 0.2253, "num_tokens": 13776190.0, "reward": -9.101318359375, "reward_std": 8.66612434387207, "rewards/rm_reward_func/mean": -9.101318359375, "rewards/rm_reward_func/std": 10.977357864379883, "step": 1046 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 369.15625, "completions/mean_terminated_length": 329.1600036621094, "completions/min_length": 95.0, "completions/min_terminated_length": 95.0, "epoch": 0.8376, "grad_norm": 14.926949501037598, "kl": 0.888671875, "learning_rate": 1e-06, "loss": -0.0087, "num_tokens": 13789795.0, "reward": 0.6767578125, "reward_std": 8.464103698730469, "rewards/rm_reward_func/mean": 0.6767578125, "rewards/rm_reward_func/std": 14.696283340454102, "step": 1047 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 358.0, "completions/mean_terminated_length": 336.0, "completions/min_length": 168.0, "completions/min_terminated_length": 168.0, "epoch": 0.8384, "grad_norm": 14.153160095214844, "kl": 3.96533203125, "learning_rate": 1e-06, "loss": 0.0663, "num_tokens": 13805603.0, "reward": -7.618194580078125, "reward_std": 5.023479461669922, "rewards/rm_reward_func/mean": -7.618194580078125, "rewards/rm_reward_func/std": 13.050457000732422, "step": 1048 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 377.9375, "completions/mean_terminated_length": 369.0000305175781, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "epoch": 0.8392, "grad_norm": 9.757351875305176, "kl": 2.5224609375, "learning_rate": 1e-06, "loss": 0.0443, "num_tokens": 13822233.0, "reward": 3.964599609375, "reward_std": 10.471548080444336, "rewards/rm_reward_func/mean": 3.964599609375, "rewards/rm_reward_func/std": 16.459672927856445, "step": 1049 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 476.0, "completions/mean_length": 310.1875, "completions/mean_terminated_length": 281.3571472167969, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "epoch": 0.84, "grad_norm": 12.101102828979492, "kl": 3.15673828125, "learning_rate": 1e-06, "loss": -0.0473, "num_tokens": 13833951.0, "reward": -2.974365234375, "reward_std": 9.494062423706055, "rewards/rm_reward_func/mean": -2.974365234375, "rewards/rm_reward_func/std": 17.267580032348633, "step": 1050 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 234.0625, "completions/mean_terminated_length": 194.35714721679688, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "epoch": 0.8408, "grad_norm": 6.150353908538818, "kl": 2.26953125, "learning_rate": 1e-06, "loss": 0.0698, "num_tokens": 13844073.0, "reward": -1.613037109375, "reward_std": 4.632699489593506, "rewards/rm_reward_func/mean": -1.613037109375, "rewards/rm_reward_func/std": 8.01832103729248, "step": 1051 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 359.6875, "completions/mean_terminated_length": 317.03997802734375, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "epoch": 0.8416, "grad_norm": 9.147581100463867, "kl": 3.4599609375, "learning_rate": 1e-06, "loss": 0.1188, "num_tokens": 13858791.0, "reward": -1.190673828125, "reward_std": 5.757209300994873, "rewards/rm_reward_func/mean": -1.190673828125, "rewards/rm_reward_func/std": 9.849881172180176, "step": 1052 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 333.71875, "completions/mean_terminated_length": 263.9565124511719, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "epoch": 0.8424, "grad_norm": 12.606836318969727, "kl": 3.59375, "learning_rate": 1e-06, "loss": 0.1426, "num_tokens": 13873742.0, "reward": -8.640029907226562, "reward_std": 7.198362350463867, "rewards/rm_reward_func/mean": -8.640029907226562, "rewards/rm_reward_func/std": 10.380491256713867, "step": 1053 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 469.0, "completions/mean_length": 230.84375, "completions/mean_terminated_length": 221.77418518066406, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.8432, "grad_norm": 6.97705078125, "kl": 2.47314453125, "learning_rate": 1e-06, "loss": 0.173, "num_tokens": 13883497.0, "reward": -2.553131103515625, "reward_std": 3.148016929626465, "rewards/rm_reward_func/mean": -2.553131103515625, "rewards/rm_reward_func/std": 8.437515258789062, "step": 1054 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 325.25, "completions/mean_terminated_length": 213.1999969482422, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "epoch": 0.844, "grad_norm": 7.593416213989258, "kl": 0.8896484375, "learning_rate": 1e-06, "loss": 0.0335, "num_tokens": 13897529.0, "reward": 6.3282470703125, "reward_std": 3.7554547786712646, "rewards/rm_reward_func/mean": 6.3282470703125, "rewards/rm_reward_func/std": 13.18265438079834, "step": 1055 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 338.25, "completions/mean_terminated_length": 289.6000061035156, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.8448, "grad_norm": 7.9972734451293945, "kl": 1.67822265625, "learning_rate": 1e-06, "loss": 0.0365, "num_tokens": 13910961.0, "reward": 9.41973876953125, "reward_std": 5.808597564697266, "rewards/rm_reward_func/mean": 9.41973876953125, "rewards/rm_reward_func/std": 13.108800888061523, "step": 1056 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 378.15625, "completions/mean_terminated_length": 317.31817626953125, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "epoch": 0.8456, "grad_norm": 9.818436622619629, "kl": 2.041015625, "learning_rate": 1e-06, "loss": 0.0809, "num_tokens": 13925710.0, "reward": 4.3310546875, "reward_std": 6.725727081298828, "rewards/rm_reward_func/mean": 4.3310546875, "rewards/rm_reward_func/std": 8.763275146484375, "step": 1057 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 465.0, "completions/mean_length": 377.5, "completions/mean_terminated_length": 352.59259033203125, "completions/min_length": 114.0, "completions/min_terminated_length": 114.0, "epoch": 0.8464, "grad_norm": 11.298351287841797, "kl": 0.86474609375, "learning_rate": 1e-06, "loss": 0.1117, "num_tokens": 13939478.0, "reward": 4.32666015625, "reward_std": 8.901585578918457, "rewards/rm_reward_func/mean": 4.32666015625, "rewards/rm_reward_func/std": 16.733640670776367, "step": 1058 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 472.0, "completions/mean_length": 346.40625, "completions/mean_terminated_length": 300.0400085449219, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "epoch": 0.8472, "grad_norm": 7.649352550506592, "kl": 2.932861328125, "learning_rate": 1e-06, "loss": -0.004, "num_tokens": 13955987.0, "reward": 10.8458251953125, "reward_std": 10.861635208129883, "rewards/rm_reward_func/mean": 10.8458251953125, "rewards/rm_reward_func/std": 20.45244598388672, "step": 1059 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 251.59375, "completions/mean_terminated_length": 203.37037658691406, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "epoch": 0.848, "grad_norm": 14.20515251159668, "kl": 1.2626953125, "learning_rate": 1e-06, "loss": -0.1202, "num_tokens": 13967622.0, "reward": 3.55889892578125, "reward_std": 3.865619659423828, "rewards/rm_reward_func/mean": 3.55889892578125, "rewards/rm_reward_func/std": 14.522747039794922, "step": 1060 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 447.0, "completions/max_terminated_length": 447.0, "completions/mean_length": 204.1875, "completions/mean_terminated_length": 204.1875, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.8488, "grad_norm": 5.686960220336914, "kl": 0.53759765625, "learning_rate": 1e-06, "loss": -0.0044, "num_tokens": 13976516.0, "reward": 10.380859375, "reward_std": 1.672519326210022, "rewards/rm_reward_func/mean": 10.380859375, "rewards/rm_reward_func/std": 4.958350658416748, "step": 1061 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 455.0, "completions/max_terminated_length": 455.0, "completions/mean_length": 294.875, "completions/mean_terminated_length": 294.875, "completions/min_length": 115.0, "completions/min_terminated_length": 115.0, "epoch": 0.8496, "grad_norm": 6.868638038635254, "kl": 0.47802734375, "learning_rate": 1e-06, "loss": -0.0463, "num_tokens": 13987656.0, "reward": -3.4047164916992188, "reward_std": 2.7297439575195312, "rewards/rm_reward_func/mean": -3.4047164916992188, "rewards/rm_reward_func/std": 8.868703842163086, "step": 1062 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 484.0, "completions/mean_length": 354.15625, "completions/mean_terminated_length": 309.9599914550781, "completions/min_length": 83.0, "completions/min_terminated_length": 83.0, "epoch": 0.8504, "grad_norm": 10.120816230773926, "kl": 0.4521484375, "learning_rate": 1e-06, "loss": 0.0789, "num_tokens": 14001477.0, "reward": 10.3330078125, "reward_std": 5.511585235595703, "rewards/rm_reward_func/mean": 10.3330078125, "rewards/rm_reward_func/std": 11.582442283630371, "step": 1063 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 383.15625, "completions/mean_terminated_length": 347.0799865722656, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.8512, "grad_norm": 8.560920715332031, "kl": 1.580078125, "learning_rate": 1e-06, "loss": 0.1501, "num_tokens": 14016578.0, "reward": 15.5166015625, "reward_std": 7.954902648925781, "rewards/rm_reward_func/mean": 15.5166015625, "rewards/rm_reward_func/std": 15.997610092163086, "step": 1064 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 219.71875, "completions/mean_terminated_length": 177.96429443359375, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.852, "grad_norm": 9.301816940307617, "kl": 0.8408203125, "learning_rate": 1e-06, "loss": 0.0325, "num_tokens": 14026161.0, "reward": 0.931640625, "reward_std": 2.5424585342407227, "rewards/rm_reward_func/mean": 0.931640625, "rewards/rm_reward_func/std": 9.632967948913574, "step": 1065 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 511.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 292.34375, "completions/mean_terminated_length": 292.34375, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.8528, "grad_norm": 12.27116584777832, "kl": 1.8857421875, "learning_rate": 1e-06, "loss": -0.0008, "num_tokens": 14038460.0, "reward": 5.19140625, "reward_std": 2.821535110473633, "rewards/rm_reward_func/mean": 5.19140625, "rewards/rm_reward_func/std": 11.627610206604004, "step": 1066 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 349.5625, "completions/mean_terminated_length": 344.32257080078125, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "epoch": 0.8536, "grad_norm": 11.470945358276367, "kl": 0.71337890625, "learning_rate": 1e-06, "loss": -0.0169, "num_tokens": 14051606.0, "reward": 3.7967529296875, "reward_std": 6.249090194702148, "rewards/rm_reward_func/mean": 3.7967529296875, "rewards/rm_reward_func/std": 6.5110979080200195, "step": 1067 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 312.90625, "completions/mean_terminated_length": 299.63336181640625, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.8544, "grad_norm": 10.407036781311035, "kl": 0.72705078125, "learning_rate": 1e-06, "loss": 0.0446, "num_tokens": 14063891.0, "reward": 16.3878173828125, "reward_std": 8.335765838623047, "rewards/rm_reward_func/mean": 16.3878173828125, "rewards/rm_reward_func/std": 16.84733009338379, "step": 1068 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 281.96875, "completions/mean_terminated_length": 266.63336181640625, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.8552, "grad_norm": 7.953994274139404, "kl": 0.6884765625, "learning_rate": 1e-06, "loss": 0.0595, "num_tokens": 14074970.0, "reward": 3.028076171875, "reward_std": 4.655251979827881, "rewards/rm_reward_func/mean": 3.028076171875, "rewards/rm_reward_func/std": 14.149062156677246, "step": 1069 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 479.0, "completions/mean_length": 300.375, "completions/mean_terminated_length": 261.1851806640625, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.856, "grad_norm": 5.069062232971191, "kl": 0.404541015625, "learning_rate": 1e-06, "loss": 0.0521, "num_tokens": 14088094.0, "reward": 1.788726806640625, "reward_std": 5.874140739440918, "rewards/rm_reward_func/mean": 1.788726806640625, "rewards/rm_reward_func/std": 10.132195472717285, "step": 1070 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 198.875, "completions/mean_terminated_length": 188.77418518066406, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "epoch": 0.8568, "grad_norm": 11.527831077575684, "kl": 1.47314453125, "learning_rate": 1e-06, "loss": 0.1007, "num_tokens": 14097274.0, "reward": -1.9715576171875, "reward_std": 2.8827223777770996, "rewards/rm_reward_func/mean": -1.9715576171875, "rewards/rm_reward_func/std": 5.209477424621582, "step": 1071 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 449.09375, "completions/mean_terminated_length": 411.3500061035156, "completions/min_length": 336.0, "completions/min_terminated_length": 336.0, "epoch": 0.8576, "grad_norm": 6.973728656768799, "kl": 0.34130859375, "learning_rate": 1e-06, "loss": 0.012, "num_tokens": 14115125.0, "reward": 8.21722412109375, "reward_std": 5.218785285949707, "rewards/rm_reward_func/mean": 8.21722412109375, "rewards/rm_reward_func/std": 12.76237964630127, "step": 1072 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 368.53125, "completions/mean_terminated_length": 312.39129638671875, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.8584, "grad_norm": 6.441647052764893, "kl": 0.60107421875, "learning_rate": 1e-06, "loss": 0.0165, "num_tokens": 14129646.0, "reward": 8.17431640625, "reward_std": 4.927266597747803, "rewards/rm_reward_func/mean": 8.17431640625, "rewards/rm_reward_func/std": 10.222173690795898, "step": 1073 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 328.34375, "completions/mean_terminated_length": 267.125, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.8592, "grad_norm": 5.157811164855957, "kl": 1.52880859375, "learning_rate": 1e-06, "loss": 0.0403, "num_tokens": 14142593.0, "reward": 3.81683349609375, "reward_std": 3.48171329498291, "rewards/rm_reward_func/mean": 3.81683349609375, "rewards/rm_reward_func/std": 9.174951553344727, "step": 1074 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 347.3125, "completions/mean_terminated_length": 272.4545593261719, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.86, "grad_norm": 5.949108600616455, "kl": 0.45263671875, "learning_rate": 1e-06, "loss": 0.0243, "num_tokens": 14155811.0, "reward": 14.94580078125, "reward_std": 4.586613655090332, "rewards/rm_reward_func/mean": 14.94580078125, "rewards/rm_reward_func/std": 6.567379951477051, "step": 1075 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 350.25, "completions/mean_terminated_length": 312.923095703125, "completions/min_length": 152.0, "completions/min_terminated_length": 152.0, "epoch": 0.8608, "grad_norm": 7.930327415466309, "kl": 0.419921875, "learning_rate": 1e-06, "loss": -0.0041, "num_tokens": 14168835.0, "reward": 2.20947265625, "reward_std": 6.557795524597168, "rewards/rm_reward_func/mean": 2.20947265625, "rewards/rm_reward_func/std": 8.4446382522583, "step": 1076 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 386.21875, "completions/mean_terminated_length": 344.29168701171875, "completions/min_length": 180.0, "completions/min_terminated_length": 180.0, "epoch": 0.8616, "grad_norm": 8.628700256347656, "kl": 0.5888671875, "learning_rate": 1e-06, "loss": -0.0536, "num_tokens": 14183226.0, "reward": 10.73797607421875, "reward_std": 5.025568008422852, "rewards/rm_reward_func/mean": 10.73797607421875, "rewards/rm_reward_func/std": 8.780284881591797, "step": 1077 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 453.21875, "completions/mean_terminated_length": 426.5, "completions/min_length": 354.0, "completions/min_terminated_length": 354.0, "epoch": 0.8624, "grad_norm": 6.511486053466797, "kl": 0.40283203125, "learning_rate": 1e-06, "loss": 0.033, "num_tokens": 14200257.0, "reward": 15.1558837890625, "reward_std": 4.903788089752197, "rewards/rm_reward_func/mean": 15.1558837890625, "rewards/rm_reward_func/std": 12.219951629638672, "step": 1078 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 492.0, "completions/mean_length": 311.78125, "completions/mean_terminated_length": 305.32257080078125, "completions/min_length": 177.0, "completions/min_terminated_length": 177.0, "epoch": 0.8632, "grad_norm": 9.137861251831055, "kl": 1.89208984375, "learning_rate": 1e-06, "loss": 0.0506, "num_tokens": 14212794.0, "reward": 3.208251953125, "reward_std": 6.694586753845215, "rewards/rm_reward_func/mean": 3.208251953125, "rewards/rm_reward_func/std": 14.088113784790039, "step": 1079 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 492.0, "completions/mean_length": 248.1875, "completions/mean_terminated_length": 230.60000610351562, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "epoch": 0.864, "grad_norm": 8.597902297973633, "kl": 3.4228515625, "learning_rate": 1e-06, "loss": 0.1484, "num_tokens": 14223552.0, "reward": -1.8214111328125, "reward_std": 4.2023773193359375, "rewards/rm_reward_func/mean": -1.8214111328125, "rewards/rm_reward_func/std": 9.527159690856934, "step": 1080 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 369.84375, "completions/mean_terminated_length": 272.5789489746094, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.8648, "grad_norm": 5.029150485992432, "kl": 0.5234375, "learning_rate": 1e-06, "loss": 0.0322, "num_tokens": 14238843.0, "reward": -7.0312652587890625, "reward_std": 2.070448637008667, "rewards/rm_reward_func/mean": -7.0312652587890625, "rewards/rm_reward_func/std": 11.313837051391602, "step": 1081 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 444.0, "completions/max_terminated_length": 444.0, "completions/mean_length": 282.59375, "completions/mean_terminated_length": 282.59375, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "epoch": 0.8656, "grad_norm": 16.06825828552246, "kl": 5.65869140625, "learning_rate": 1e-06, "loss": 0.2783, "num_tokens": 14251942.0, "reward": 3.9051666259765625, "reward_std": 4.887936592102051, "rewards/rm_reward_func/mean": 3.9051666259765625, "rewards/rm_reward_func/std": 14.745894432067871, "step": 1082 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 458.0, "completions/mean_length": 306.90625, "completions/mean_terminated_length": 300.2903137207031, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "epoch": 0.8664, "grad_norm": 7.322081089019775, "kl": 3.85546875, "learning_rate": 1e-06, "loss": 0.1572, "num_tokens": 14267451.0, "reward": 8.35272216796875, "reward_std": 9.634117126464844, "rewards/rm_reward_func/mean": 8.35272216796875, "rewards/rm_reward_func/std": 19.10032844543457, "step": 1083 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 345.5, "completions/mean_terminated_length": 340.1290283203125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "epoch": 0.8672, "grad_norm": 12.525562286376953, "kl": 5.53955078125, "learning_rate": 1e-06, "loss": 0.4021, "num_tokens": 14285251.0, "reward": 14.5098876953125, "reward_std": 6.451245307922363, "rewards/rm_reward_func/mean": 14.5098876953125, "rewards/rm_reward_func/std": 12.280783653259277, "step": 1084 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 491.0, "completions/mean_length": 227.1875, "completions/mean_terminated_length": 197.72413635253906, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "epoch": 0.868, "grad_norm": 12.410992622375488, "kl": 6.01171875, "learning_rate": 1e-06, "loss": 0.2138, "num_tokens": 14295113.0, "reward": 2.096435546875, "reward_std": 6.907596588134766, "rewards/rm_reward_func/mean": 2.096435546875, "rewards/rm_reward_func/std": 11.79339599609375, "step": 1085 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 475.0, "completions/mean_length": 263.28125, "completions/mean_terminated_length": 246.70001220703125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "epoch": 0.8688, "grad_norm": 7.831367015838623, "kl": 0.89599609375, "learning_rate": 1e-06, "loss": -0.026, "num_tokens": 14308474.0, "reward": 2.3360595703125, "reward_std": 6.861347675323486, "rewards/rm_reward_func/mean": 2.3360595703125, "rewards/rm_reward_func/std": 9.401144027709961, "step": 1086 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 246.40625, "completions/mean_terminated_length": 237.8386993408203, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "epoch": 0.8696, "grad_norm": 12.294774055480957, "kl": 3.60009765625, "learning_rate": 1e-06, "loss": 0.0805, "num_tokens": 14319327.0, "reward": 1.0987548828125, "reward_std": 5.113190650939941, "rewards/rm_reward_func/mean": 1.0987548828125, "rewards/rm_reward_func/std": 7.818641185760498, "step": 1087 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 479.0, "completions/mean_length": 269.0, "completions/mean_terminated_length": 261.1612854003906, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "epoch": 0.8704, "grad_norm": 13.56505298614502, "kl": 4.18359375, "learning_rate": 1e-06, "loss": 0.114, "num_tokens": 14331231.0, "reward": 1.3173828125, "reward_std": 7.998728275299072, "rewards/rm_reward_func/mean": 1.3173828125, "rewards/rm_reward_func/std": 18.137277603149414, "step": 1088 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 472.0, "completions/mean_length": 299.0625, "completions/mean_terminated_length": 249.92308044433594, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.8712, "grad_norm": 5.182938575744629, "kl": 1.154296875, "learning_rate": 1e-06, "loss": 0.0724, "num_tokens": 14343945.0, "reward": 9.21612548828125, "reward_std": 8.317728042602539, "rewards/rm_reward_func/mean": 9.21612548828125, "rewards/rm_reward_func/std": 16.426794052124023, "step": 1089 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 403.0, "completions/max_terminated_length": 403.0, "completions/mean_length": 185.0, "completions/mean_terminated_length": 185.0, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "epoch": 0.872, "grad_norm": 68.98828125, "kl": 4.005859375, "learning_rate": 1e-06, "loss": 0.2527, "num_tokens": 14354649.0, "reward": -9.63653564453125, "reward_std": 3.627918243408203, "rewards/rm_reward_func/mean": -9.63653564453125, "rewards/rm_reward_func/std": 8.9398775100708, "step": 1090 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 488.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 313.59375, "completions/mean_terminated_length": 313.59375, "completions/min_length": 111.0, "completions/min_terminated_length": 111.0, "epoch": 0.8728, "grad_norm": 13.77872085571289, "kl": 5.103515625, "learning_rate": 1e-06, "loss": 0.2633, "num_tokens": 14367212.0, "reward": 0.75341796875, "reward_std": 5.394412040710449, "rewards/rm_reward_func/mean": 0.75341796875, "rewards/rm_reward_func/std": 13.423392295837402, "step": 1091 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 239.0, "completions/mean_length": 228.75, "completions/mean_terminated_length": 134.33334350585938, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.8736, "grad_norm": 11.227128028869629, "kl": 1.6025390625, "learning_rate": 1e-06, "loss": 0.1076, "num_tokens": 14378772.0, "reward": -0.025146484375, "reward_std": 2.094247341156006, "rewards/rm_reward_func/mean": -0.025146484375, "rewards/rm_reward_func/std": 8.626964569091797, "step": 1092 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 347.5, "completions/mean_terminated_length": 283.13043212890625, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.8744, "grad_norm": 6.2191619873046875, "kl": 3.33203125, "learning_rate": 1e-06, "loss": 0.1357, "num_tokens": 14394876.0, "reward": 16.056884765625, "reward_std": 9.725163459777832, "rewards/rm_reward_func/mean": 16.056884765625, "rewards/rm_reward_func/std": 21.10055923461914, "step": 1093 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 407.0, "completions/max_terminated_length": 407.0, "completions/mean_length": 238.3125, "completions/mean_terminated_length": 238.3125, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "epoch": 0.8752, "grad_norm": 9.5230073928833, "kl": 3.8193359375, "learning_rate": 1e-06, "loss": 0.0945, "num_tokens": 14404414.0, "reward": -4.79345703125, "reward_std": 5.990765571594238, "rewards/rm_reward_func/mean": -4.79345703125, "rewards/rm_reward_func/std": 13.47055721282959, "step": 1094 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 236.90625, "completions/mean_terminated_length": 185.9629669189453, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "epoch": 0.876, "grad_norm": 7.131078243255615, "kl": 1.12109375, "learning_rate": 1e-06, "loss": 0.0133, "num_tokens": 14415475.0, "reward": 0.505615234375, "reward_std": 3.531787395477295, "rewards/rm_reward_func/mean": 0.505615234375, "rewards/rm_reward_func/std": 6.195098876953125, "step": 1095 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 386.0, "completions/max_terminated_length": 386.0, "completions/mean_length": 217.75, "completions/mean_terminated_length": 217.75, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.8768, "grad_norm": 7.909564971923828, "kl": 1.53125, "learning_rate": 1e-06, "loss": 0.0247, "num_tokens": 14425171.0, "reward": -1.14208984375, "reward_std": 1.705171823501587, "rewards/rm_reward_func/mean": -1.14208984375, "rewards/rm_reward_func/std": 10.815631866455078, "step": 1096 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 265.15625, "completions/mean_terminated_length": 182.875, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "epoch": 0.8776, "grad_norm": 4.670645236968994, "kl": 2.76171875, "learning_rate": 1e-06, "loss": 0.1654, "num_tokens": 14438568.0, "reward": 10.848876953125, "reward_std": 4.882187366485596, "rewards/rm_reward_func/mean": 10.848876953125, "rewards/rm_reward_func/std": 13.137846946716309, "step": 1097 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 390.0, "completions/mean_length": 317.75, "completions/mean_terminated_length": 253.0, "completions/min_length": 124.0, "completions/min_terminated_length": 124.0, "epoch": 0.8784, "grad_norm": 11.463107109069824, "kl": 3.96484375, "learning_rate": 1e-06, "loss": 0.1702, "num_tokens": 14451368.0, "reward": 4.6419677734375, "reward_std": 3.999204397201538, "rewards/rm_reward_func/mean": 4.6419677734375, "rewards/rm_reward_func/std": 16.210906982421875, "step": 1098 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 488.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 302.15625, "completions/mean_terminated_length": 302.15625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "epoch": 0.8792, "grad_norm": 21.74379539489746, "kl": 7.875, "learning_rate": 1e-06, "loss": 0.225, "num_tokens": 14463477.0, "reward": -5.63720703125, "reward_std": 9.232072830200195, "rewards/rm_reward_func/mean": -5.63720703125, "rewards/rm_reward_func/std": 18.789886474609375, "step": 1099 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 299.25, "completions/mean_terminated_length": 259.85186767578125, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.88, "grad_norm": 6.47990608215332, "kl": 5.8642578125, "learning_rate": 1e-06, "loss": 0.1774, "num_tokens": 14476309.0, "reward": -0.43701171875, "reward_std": 10.200658798217773, "rewards/rm_reward_func/mean": -0.43701171875, "rewards/rm_reward_func/std": 14.860036849975586, "step": 1100 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 499.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 105.0, "completions/mean_terminated_length": 105.0, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "epoch": 0.8808, "grad_norm": 5.006459712982178, "kl": 1.625, "learning_rate": 1e-06, "loss": -0.0308, "num_tokens": 14483213.0, "reward": 1.8359375, "reward_std": 0.8900178074836731, "rewards/rm_reward_func/mean": 1.8359375, "rewards/rm_reward_func/std": 11.928114891052246, "step": 1101 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 280.28125, "completions/mean_terminated_length": 264.8333435058594, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.8816, "grad_norm": 8.10487174987793, "kl": 2.853515625, "learning_rate": 1e-06, "loss": 0.0011, "num_tokens": 14495190.0, "reward": 5.513671875, "reward_std": 8.272500991821289, "rewards/rm_reward_func/mean": 5.513671875, "rewards/rm_reward_func/std": 16.234622955322266, "step": 1102 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 239.5, "completions/mean_terminated_length": 230.7096710205078, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.8824, "grad_norm": 3.8931567668914795, "kl": 0.5224609375, "learning_rate": 1e-06, "loss": 0.0255, "num_tokens": 14505822.0, "reward": 11.3662109375, "reward_std": 2.4082765579223633, "rewards/rm_reward_func/mean": 11.3662109375, "rewards/rm_reward_func/std": 5.015256404876709, "step": 1103 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 492.0, "completions/mean_length": 237.28125, "completions/mean_terminated_length": 198.0357208251953, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.8832, "grad_norm": 10.963964462280273, "kl": 4.1640625, "learning_rate": 1e-06, "loss": 0.1059, "num_tokens": 14518447.0, "reward": 1.65234375, "reward_std": 3.4185004234313965, "rewards/rm_reward_func/mean": 1.65234375, "rewards/rm_reward_func/std": 13.96972370147705, "step": 1104 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 469.0, "completions/max_terminated_length": 469.0, "completions/mean_length": 252.53125, "completions/mean_terminated_length": 252.53125, "completions/min_length": 83.0, "completions/min_terminated_length": 83.0, "epoch": 0.884, "grad_norm": 7.281085014343262, "kl": 4.080078125, "learning_rate": 1e-06, "loss": 0.212, "num_tokens": 14529264.0, "reward": -5.43353271484375, "reward_std": 5.849520206451416, "rewards/rm_reward_func/mean": -5.43353271484375, "rewards/rm_reward_func/std": 9.325867652893066, "step": 1105 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 492.0, "completions/max_terminated_length": 492.0, "completions/mean_length": 341.4375, "completions/mean_terminated_length": 341.4375, "completions/min_length": 202.0, "completions/min_terminated_length": 202.0, "epoch": 0.8848, "grad_norm": 7.2115478515625, "kl": 0.4150390625, "learning_rate": 1e-06, "loss": 0.0024, "num_tokens": 14543294.0, "reward": 7.579345703125, "reward_std": 4.300886154174805, "rewards/rm_reward_func/mean": 7.579345703125, "rewards/rm_reward_func/std": 5.634270191192627, "step": 1106 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 450.0, "completions/max_terminated_length": 450.0, "completions/mean_length": 281.21875, "completions/mean_terminated_length": 281.21875, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.8856, "grad_norm": 11.654593467712402, "kl": 1.85693359375, "learning_rate": 1e-06, "loss": 0.0569, "num_tokens": 14554965.0, "reward": 2.6845703125, "reward_std": 5.363699436187744, "rewards/rm_reward_func/mean": 2.6845703125, "rewards/rm_reward_func/std": 11.830452919006348, "step": 1107 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 509.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 215.78125, "completions/mean_terminated_length": 215.78125, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.8864, "grad_norm": 7.386474132537842, "kl": 0.47412109375, "learning_rate": 1e-06, "loss": -0.0309, "num_tokens": 14564774.0, "reward": 16.75201416015625, "reward_std": 2.713106155395508, "rewards/rm_reward_func/mean": 16.75201416015625, "rewards/rm_reward_func/std": 14.134072303771973, "step": 1108 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 278.9375, "completions/mean_terminated_length": 245.6428680419922, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.8872, "grad_norm": 8.896984100341797, "kl": 0.71337890625, "learning_rate": 1e-06, "loss": -0.012, "num_tokens": 14576156.0, "reward": 3.846038818359375, "reward_std": 4.625146389007568, "rewards/rm_reward_func/mean": 3.846038818359375, "rewards/rm_reward_func/std": 7.2643961906433105, "step": 1109 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 401.84375, "completions/mean_terminated_length": 386.1071472167969, "completions/min_length": 247.0, "completions/min_terminated_length": 247.0, "epoch": 0.888, "grad_norm": 9.240114212036133, "kl": 1.05859375, "learning_rate": 1e-06, "loss": 0.0088, "num_tokens": 14591575.0, "reward": 8.68145751953125, "reward_std": 6.467904567718506, "rewards/rm_reward_func/mean": 8.68145751953125, "rewards/rm_reward_func/std": 9.515111923217773, "step": 1110 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 397.84375, "completions/mean_terminated_length": 345.9545593261719, "completions/min_length": 235.0, "completions/min_terminated_length": 235.0, "epoch": 0.8888, "grad_norm": 6.770890235900879, "kl": 2.41015625, "learning_rate": 1e-06, "loss": 0.0554, "num_tokens": 14606562.0, "reward": -2.921142578125, "reward_std": 5.699788570404053, "rewards/rm_reward_func/mean": -2.921142578125, "rewards/rm_reward_func/std": 17.231367111206055, "step": 1111 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 480.0, "completions/mean_length": 187.0, "completions/mean_terminated_length": 96.0, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.8896, "grad_norm": 4.088595390319824, "kl": 0.5556640625, "learning_rate": 1e-06, "loss": 0.025, "num_tokens": 14618274.0, "reward": 4.689453125, "reward_std": 0.7515532970428467, "rewards/rm_reward_func/mean": 4.689453125, "rewards/rm_reward_func/std": 11.583274841308594, "step": 1112 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 283.46875, "completions/mean_terminated_length": 179.59091186523438, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.8904, "grad_norm": 5.539703369140625, "kl": 2.7138671875, "learning_rate": 1e-06, "loss": 0.1386, "num_tokens": 14631761.0, "reward": 2.4609222412109375, "reward_std": 4.746338844299316, "rewards/rm_reward_func/mean": 2.4609222412109375, "rewards/rm_reward_func/std": 9.370748519897461, "step": 1113 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 264.59375, "completions/mean_terminated_length": 248.10000610351562, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "epoch": 0.8912, "grad_norm": 16.168067932128906, "kl": 1.09033203125, "learning_rate": 1e-06, "loss": -0.1252, "num_tokens": 14643388.0, "reward": 1.837890625, "reward_std": 7.579433441162109, "rewards/rm_reward_func/mean": 1.837890625, "rewards/rm_reward_func/std": 13.530681610107422, "step": 1114 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 415.03125, "completions/mean_terminated_length": 364.23809814453125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "epoch": 0.892, "grad_norm": 15.032100677490234, "kl": 4.005859375, "learning_rate": 1e-06, "loss": 0.101, "num_tokens": 14662877.0, "reward": -12.3818359375, "reward_std": 5.061200141906738, "rewards/rm_reward_func/mean": -12.3818359375, "rewards/rm_reward_func/std": 8.946840286254883, "step": 1115 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 241.1875, "completions/mean_terminated_length": 213.1724090576172, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "epoch": 0.8928, "grad_norm": 12.395576477050781, "kl": 3.50927734375, "learning_rate": 1e-06, "loss": 0.3018, "num_tokens": 14673483.0, "reward": -4.934326171875, "reward_std": 6.060213088989258, "rewards/rm_reward_func/mean": -4.934326171875, "rewards/rm_reward_func/std": 12.644729614257812, "step": 1116 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 316.15625, "completions/mean_terminated_length": 250.875, "completions/min_length": 107.0, "completions/min_terminated_length": 107.0, "epoch": 0.8936, "grad_norm": 9.258816719055176, "kl": 4.6328125, "learning_rate": 1e-06, "loss": 0.1356, "num_tokens": 14685808.0, "reward": -9.6767578125, "reward_std": 4.323359489440918, "rewards/rm_reward_func/mean": -9.6767578125, "rewards/rm_reward_func/std": 4.797229290008545, "step": 1117 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 347.75, "completions/mean_terminated_length": 336.8000183105469, "completions/min_length": 150.0, "completions/min_terminated_length": 150.0, "epoch": 0.8944, "grad_norm": 13.36823558807373, "kl": 3.09423828125, "learning_rate": 1e-06, "loss": 0.0864, "num_tokens": 14698928.0, "reward": 3.083251953125, "reward_std": 6.492587089538574, "rewards/rm_reward_func/mean": 3.083251953125, "rewards/rm_reward_func/std": 15.465557098388672, "step": 1118 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 344.90625, "completions/mean_terminated_length": 327.6206970214844, "completions/min_length": 94.0, "completions/min_terminated_length": 94.0, "epoch": 0.8952, "grad_norm": 16.696882247924805, "kl": 1.796875, "learning_rate": 1e-06, "loss": 0.026, "num_tokens": 14714893.0, "reward": 4.2606201171875, "reward_std": 6.969727516174316, "rewards/rm_reward_func/mean": 4.2606201171875, "rewards/rm_reward_func/std": 21.33582305908203, "step": 1119 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 372.0, "completions/mean_length": 211.4375, "completions/mean_terminated_length": 201.74192810058594, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.896, "grad_norm": 36.645877838134766, "kl": 3.154296875, "learning_rate": 1e-06, "loss": 0.3064, "num_tokens": 14726907.0, "reward": -5.021240234375, "reward_std": 5.98872184753418, "rewards/rm_reward_func/mean": -5.021240234375, "rewards/rm_reward_func/std": 11.189291000366211, "step": 1120 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 410.0, "completions/max_terminated_length": 410.0, "completions/mean_length": 206.65625, "completions/mean_terminated_length": 206.65625, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.8968, "grad_norm": 5.673826217651367, "kl": 0.5107421875, "learning_rate": 1e-06, "loss": -0.0006, "num_tokens": 14739032.0, "reward": 10.79296875, "reward_std": 1.6997040510177612, "rewards/rm_reward_func/mean": 10.79296875, "rewards/rm_reward_func/std": 6.225403785705566, "step": 1121 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 308.21875, "completions/mean_terminated_length": 270.4814758300781, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.8976, "grad_norm": 9.43205451965332, "kl": 4.9072265625, "learning_rate": 1e-06, "loss": 0.214, "num_tokens": 14751527.0, "reward": -3.093505859375, "reward_std": 5.163877487182617, "rewards/rm_reward_func/mean": -3.093505859375, "rewards/rm_reward_func/std": 11.953996658325195, "step": 1122 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 342.0, "completions/mean_length": 182.8125, "completions/mean_terminated_length": 135.7857208251953, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.8984, "grad_norm": 12.99282169342041, "kl": 2.005859375, "learning_rate": 1e-06, "loss": 0.2826, "num_tokens": 14761121.0, "reward": 0.334228515625, "reward_std": 4.173687934875488, "rewards/rm_reward_func/mean": 0.334228515625, "rewards/rm_reward_func/std": 6.381908416748047, "step": 1123 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 185.0625, "completions/mean_terminated_length": 124.51851654052734, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.8992, "grad_norm": 9.33836841583252, "kl": 0.52001953125, "learning_rate": 1e-06, "loss": 0.022, "num_tokens": 14770411.0, "reward": 7.060302734375, "reward_std": 1.1086238622665405, "rewards/rm_reward_func/mean": 7.060302734375, "rewards/rm_reward_func/std": 5.903942584991455, "step": 1124 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 345.09375, "completions/mean_terminated_length": 321.25, "completions/min_length": 166.0, "completions/min_terminated_length": 166.0, "epoch": 0.9, "grad_norm": 10.884320259094238, "kl": 2.587890625, "learning_rate": 1e-06, "loss": 0.0485, "num_tokens": 14786686.0, "reward": 11.99560546875, "reward_std": 7.212575912475586, "rewards/rm_reward_func/mean": 11.99560546875, "rewards/rm_reward_func/std": 22.268768310546875, "step": 1125 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 448.0, "completions/max_terminated_length": 448.0, "completions/mean_length": 232.03125, "completions/mean_terminated_length": 232.03125, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9008, "grad_norm": 9.929396629333496, "kl": 1.83203125, "learning_rate": 1e-06, "loss": 0.0425, "num_tokens": 14795951.0, "reward": -1.327392578125, "reward_std": 3.5014753341674805, "rewards/rm_reward_func/mean": -1.327392578125, "rewards/rm_reward_func/std": 10.548651695251465, "step": 1126 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 410.625, "completions/mean_terminated_length": 357.5238037109375, "completions/min_length": 252.0, "completions/min_terminated_length": 252.0, "epoch": 0.9016, "grad_norm": 10.775675773620605, "kl": 1.833740234375, "learning_rate": 1e-06, "loss": 0.0623, "num_tokens": 14812659.0, "reward": 6.20263671875, "reward_std": 6.819557189941406, "rewards/rm_reward_func/mean": 6.20263671875, "rewards/rm_reward_func/std": 21.737092971801758, "step": 1127 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.46875, "completions/max_length": 512.0, "completions/max_terminated_length": 476.0, "completions/mean_length": 370.125, "completions/mean_terminated_length": 244.94117736816406, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9024, "grad_norm": 8.568568229675293, "kl": 1.802734375, "learning_rate": 1e-06, "loss": 0.0395, "num_tokens": 14826751.0, "reward": -0.14697265625, "reward_std": 5.660521507263184, "rewards/rm_reward_func/mean": -0.14697265625, "rewards/rm_reward_func/std": 9.781434059143066, "step": 1128 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 389.375, "completions/mean_terminated_length": 333.6363830566406, "completions/min_length": 131.0, "completions/min_terminated_length": 131.0, "epoch": 0.9032, "grad_norm": 14.49055004119873, "kl": 1.3701171875, "learning_rate": 1e-06, "loss": 0.1125, "num_tokens": 14842283.0, "reward": -4.70123291015625, "reward_std": 5.61661434173584, "rewards/rm_reward_func/mean": -4.70123291015625, "rewards/rm_reward_func/std": 6.281442165374756, "step": 1129 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 377.125, "completions/mean_terminated_length": 357.8571472167969, "completions/min_length": 132.0, "completions/min_terminated_length": 132.0, "epoch": 0.904, "grad_norm": 8.899121284484863, "kl": 0.798828125, "learning_rate": 1e-06, "loss": -0.0292, "num_tokens": 14856479.0, "reward": 11.397480010986328, "reward_std": 5.290931701660156, "rewards/rm_reward_func/mean": 11.397480010986328, "rewards/rm_reward_func/std": 16.51535415649414, "step": 1130 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 404.15625, "completions/mean_terminated_length": 361.95654296875, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "epoch": 0.9048, "grad_norm": 20.565969467163086, "kl": 2.8994140625, "learning_rate": 1e-06, "loss": -0.0021, "num_tokens": 14872308.0, "reward": -4.064208984375, "reward_std": 9.170465469360352, "rewards/rm_reward_func/mean": -4.064208984375, "rewards/rm_reward_func/std": 11.234378814697266, "step": 1131 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 282.34375, "completions/mean_terminated_length": 267.0333557128906, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9056, "grad_norm": 38.7857551574707, "kl": 1.5107421875, "learning_rate": 1e-06, "loss": 0.0238, "num_tokens": 14884023.0, "reward": 4.152587890625, "reward_std": 8.475656509399414, "rewards/rm_reward_func/mean": 4.152587890625, "rewards/rm_reward_func/std": 13.332249641418457, "step": 1132 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 248.625, "completions/mean_terminated_length": 221.37930297851562, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "epoch": 0.9064, "grad_norm": 13.033285140991211, "kl": 1.998046875, "learning_rate": 1e-06, "loss": 0.0814, "num_tokens": 14894971.0, "reward": 2.286590576171875, "reward_std": 8.568852424621582, "rewards/rm_reward_func/mean": 2.286590576171875, "rewards/rm_reward_func/std": 21.08578109741211, "step": 1133 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 493.0, "completions/mean_length": 348.375, "completions/mean_terminated_length": 343.0967712402344, "completions/min_length": 141.0, "completions/min_terminated_length": 141.0, "epoch": 0.9072, "grad_norm": 13.447713851928711, "kl": 5.14453125, "learning_rate": 1e-06, "loss": 0.2624, "num_tokens": 14908463.0, "reward": -6.699335098266602, "reward_std": 6.876185417175293, "rewards/rm_reward_func/mean": -6.699335098266602, "rewards/rm_reward_func/std": 8.872479438781738, "step": 1134 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 366.78125, "completions/mean_terminated_length": 339.8888854980469, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "epoch": 0.908, "grad_norm": 28.221643447875977, "kl": 3.7958984375, "learning_rate": 1e-06, "loss": 0.0659, "num_tokens": 14922760.0, "reward": -6.485260009765625, "reward_std": 7.134948253631592, "rewards/rm_reward_func/mean": -6.485260009765625, "rewards/rm_reward_func/std": 11.682751655578613, "step": 1135 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 482.0, "completions/mean_length": 265.59375, "completions/mean_terminated_length": 208.73077392578125, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "epoch": 0.9088, "grad_norm": 101.20088195800781, "kl": 2.15234375, "learning_rate": 1e-06, "loss": -0.1044, "num_tokens": 14937771.0, "reward": 0.34814453125, "reward_std": 8.503646850585938, "rewards/rm_reward_func/mean": 0.34814453125, "rewards/rm_reward_func/std": 12.46761703491211, "step": 1136 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 467.0, "completions/mean_length": 288.5625, "completions/mean_terminated_length": 273.66668701171875, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9096, "grad_norm": 10.237065315246582, "kl": 1.3642578125, "learning_rate": 1e-06, "loss": 0.0369, "num_tokens": 14951197.0, "reward": 0.99951171875, "reward_std": 5.672191143035889, "rewards/rm_reward_func/mean": 0.99951171875, "rewards/rm_reward_func/std": 10.137231826782227, "step": 1137 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 492.0, "completions/mean_length": 261.21875, "completions/mean_terminated_length": 225.3928680419922, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "epoch": 0.9104, "grad_norm": 19.81197166442871, "kl": 4.25390625, "learning_rate": 1e-06, "loss": 0.088, "num_tokens": 14961988.0, "reward": -4.6865234375, "reward_std": 8.24875259399414, "rewards/rm_reward_func/mean": -4.6865234375, "rewards/rm_reward_func/std": 21.321937561035156, "step": 1138 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 308.875, "completions/mean_terminated_length": 287.862060546875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "epoch": 0.9112, "grad_norm": 17.55968475341797, "kl": 4.6123046875, "learning_rate": 1e-06, "loss": 0.1737, "num_tokens": 14974224.0, "reward": -4.378509521484375, "reward_std": 5.992432594299316, "rewards/rm_reward_func/mean": -4.378509521484375, "rewards/rm_reward_func/std": 11.200332641601562, "step": 1139 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.53125, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 454.5, "completions/mean_terminated_length": 389.3333435058594, "completions/min_length": 162.0, "completions/min_terminated_length": 162.0, "epoch": 0.912, "grad_norm": 86.18330383300781, "kl": 3.326171875, "learning_rate": 1e-06, "loss": 0.0577, "num_tokens": 14992072.0, "reward": -0.056640625, "reward_std": 9.960199356079102, "rewards/rm_reward_func/mean": -0.056640625, "rewards/rm_reward_func/std": 13.414643287658691, "step": 1140 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 504.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 271.09375, "completions/mean_terminated_length": 271.09375, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9128, "grad_norm": 25.556337356567383, "kl": 5.330078125, "learning_rate": 1e-06, "loss": 0.2298, "num_tokens": 15002867.0, "reward": -7.34228515625, "reward_std": 4.103965759277344, "rewards/rm_reward_func/mean": -7.34228515625, "rewards/rm_reward_func/std": 8.186371803283691, "step": 1141 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 479.0, "completions/max_terminated_length": 479.0, "completions/mean_length": 199.3125, "completions/mean_terminated_length": 199.3125, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "epoch": 0.9136, "grad_norm": 33.93816375732422, "kl": 5.087890625, "learning_rate": 1e-06, "loss": 0.1551, "num_tokens": 15011485.0, "reward": -7.401123046875, "reward_std": 7.4005818367004395, "rewards/rm_reward_func/mean": -7.401123046875, "rewards/rm_reward_func/std": 17.179683685302734, "step": 1142 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 463.0, "completions/mean_length": 250.84375, "completions/mean_terminated_length": 190.57693481445312, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9144, "grad_norm": 22.07563591003418, "kl": 4.1875, "learning_rate": 1e-06, "loss": 0.069, "num_tokens": 15025976.0, "reward": 0.76611328125, "reward_std": 4.296019077301025, "rewards/rm_reward_func/mean": 0.76611328125, "rewards/rm_reward_func/std": 12.173991203308105, "step": 1143 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 435.0, "completions/max_terminated_length": 435.0, "completions/mean_length": 278.59375, "completions/mean_terminated_length": 278.59375, "completions/min_length": 89.0, "completions/min_terminated_length": 89.0, "epoch": 0.9152, "grad_norm": 22.454143524169922, "kl": 2.92822265625, "learning_rate": 1e-06, "loss": -0.0074, "num_tokens": 15041539.0, "reward": -3.9213485717773438, "reward_std": 7.493886470794678, "rewards/rm_reward_func/mean": -3.9213485717773438, "rewards/rm_reward_func/std": 12.379457473754883, "step": 1144 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 385.40625, "completions/mean_terminated_length": 343.2083435058594, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "epoch": 0.916, "grad_norm": 27.009363174438477, "kl": 4.2822265625, "learning_rate": 1e-06, "loss": 0.1094, "num_tokens": 15056096.0, "reward": -11.80712890625, "reward_std": 8.341408729553223, "rewards/rm_reward_func/mean": -11.80712890625, "rewards/rm_reward_func/std": 11.495277404785156, "step": 1145 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 479.0, "completions/mean_length": 266.65625, "completions/mean_terminated_length": 258.7419128417969, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9168, "grad_norm": 43.986976623535156, "kl": 1.9150390625, "learning_rate": 1e-06, "loss": -0.0023, "num_tokens": 15066469.0, "reward": 4.2705078125, "reward_std": 8.840106964111328, "rewards/rm_reward_func/mean": 4.2705078125, "rewards/rm_reward_func/std": 16.68376922607422, "step": 1146 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 482.0, "completions/mean_length": 333.5625, "completions/mean_terminated_length": 300.5185241699219, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "epoch": 0.9176, "grad_norm": 10.10371208190918, "kl": 1.15283203125, "learning_rate": 1e-06, "loss": -0.0467, "num_tokens": 15084015.0, "reward": 10.65411376953125, "reward_std": 9.344249725341797, "rewards/rm_reward_func/mean": 10.65411376953125, "rewards/rm_reward_func/std": 16.203916549682617, "step": 1147 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 311.8125, "completions/mean_terminated_length": 274.7407531738281, "completions/min_length": 82.0, "completions/min_terminated_length": 82.0, "epoch": 0.9184, "grad_norm": 36.31003952026367, "kl": 3.45703125, "learning_rate": 1e-06, "loss": 0.0783, "num_tokens": 15096569.0, "reward": -10.46044921875, "reward_std": 8.902938842773438, "rewards/rm_reward_func/mean": -10.46044921875, "rewards/rm_reward_func/std": 14.822985649108887, "step": 1148 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40625, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 440.59375, "completions/mean_terminated_length": 391.7368469238281, "completions/min_length": 273.0, "completions/min_terminated_length": 273.0, "epoch": 0.9192, "grad_norm": 14.747289657592773, "kl": 1.51904296875, "learning_rate": 1e-06, "loss": 0.0563, "num_tokens": 15113956.0, "reward": 6.316192626953125, "reward_std": 6.612483501434326, "rewards/rm_reward_func/mean": 6.316192626953125, "rewards/rm_reward_func/std": 10.599157333374023, "step": 1149 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 488.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 261.78125, "completions/mean_terminated_length": 261.78125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "epoch": 0.92, "grad_norm": 21.25505828857422, "kl": 3.0498046875, "learning_rate": 1e-06, "loss": 0.0598, "num_tokens": 15124293.0, "reward": -5.06103515625, "reward_std": 5.579965591430664, "rewards/rm_reward_func/mean": -5.06103515625, "rewards/rm_reward_func/std": 10.378743171691895, "step": 1150 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 484.0, "completions/mean_length": 267.84375, "completions/mean_terminated_length": 222.629638671875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "epoch": 0.9208, "grad_norm": 25.507362365722656, "kl": 0.8896484375, "learning_rate": 1e-06, "loss": -0.1392, "num_tokens": 15135672.0, "reward": 1.32666015625, "reward_std": 8.384624481201172, "rewards/rm_reward_func/mean": 1.32666015625, "rewards/rm_reward_func/std": 11.488581657409668, "step": 1151 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 356.6875, "completions/mean_terminated_length": 263.5, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9216, "grad_norm": 9.581064224243164, "kl": 1.3369140625, "learning_rate": 1e-06, "loss": 0.037, "num_tokens": 15150006.0, "reward": -1.42645263671875, "reward_std": 4.21787691116333, "rewards/rm_reward_func/mean": -1.42645263671875, "rewards/rm_reward_func/std": 8.071844100952148, "step": 1152 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 506.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 222.625, "completions/mean_terminated_length": 222.625, "completions/min_length": 118.0, "completions/min_terminated_length": 118.0, "epoch": 0.9224, "grad_norm": 18.28358268737793, "kl": 0.68798828125, "learning_rate": 1e-06, "loss": 0.0422, "num_tokens": 15159922.0, "reward": -3.984771728515625, "reward_std": 4.933001518249512, "rewards/rm_reward_func/mean": -3.984771728515625, "rewards/rm_reward_func/std": 5.989495277404785, "step": 1153 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 261.90625, "completions/mean_terminated_length": 253.8386993408203, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9232, "grad_norm": 16.853899002075195, "kl": 1.552734375, "learning_rate": 1e-06, "loss": -0.0088, "num_tokens": 15170879.0, "reward": 4.74188232421875, "reward_std": 8.369260787963867, "rewards/rm_reward_func/mean": 4.74188232421875, "rewards/rm_reward_func/std": 15.880949974060059, "step": 1154 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 277.875, "completions/mean_terminated_length": 223.84616088867188, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.924, "grad_norm": 8.054681777954102, "kl": 1.3564453125, "learning_rate": 1e-06, "loss": 0.0452, "num_tokens": 15182435.0, "reward": -0.05224609375, "reward_std": 5.243035316467285, "rewards/rm_reward_func/mean": -0.05224609375, "rewards/rm_reward_func/std": 11.808974266052246, "step": 1155 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 370.6875, "completions/mean_terminated_length": 331.1199951171875, "completions/min_length": 148.0, "completions/min_terminated_length": 148.0, "epoch": 0.9248, "grad_norm": 12.995421409606934, "kl": 0.4638671875, "learning_rate": 1e-06, "loss": 0.075, "num_tokens": 15196273.0, "reward": -0.77685546875, "reward_std": 3.534580945968628, "rewards/rm_reward_func/mean": -0.77685546875, "rewards/rm_reward_func/std": 10.869222640991211, "step": 1156 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 375.25, "completions/mean_terminated_length": 303.6190490722656, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "epoch": 0.9256, "grad_norm": 29.573589324951172, "kl": 2.072265625, "learning_rate": 1e-06, "loss": 0.0377, "num_tokens": 15210377.0, "reward": 3.455078125, "reward_std": 9.187593460083008, "rewards/rm_reward_func/mean": 3.455078125, "rewards/rm_reward_func/std": 15.553760528564453, "step": 1157 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 482.0, "completions/mean_length": 393.75, "completions/mean_terminated_length": 347.478271484375, "completions/min_length": 149.0, "completions/min_terminated_length": 149.0, "epoch": 0.9264, "grad_norm": 10.407255172729492, "kl": 1.662109375, "learning_rate": 1e-06, "loss": 0.0592, "num_tokens": 15225153.0, "reward": -4.851356506347656, "reward_std": 8.531854629516602, "rewards/rm_reward_func/mean": -4.851356506347656, "rewards/rm_reward_func/std": 12.067245483398438, "step": 1158 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 483.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 228.5, "completions/mean_terminated_length": 228.5, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9272, "grad_norm": 22.906204223632812, "kl": 0.4560546875, "learning_rate": 1e-06, "loss": -0.2539, "num_tokens": 15236201.0, "reward": 4.116214752197266, "reward_std": 7.407968521118164, "rewards/rm_reward_func/mean": 4.116214752197266, "rewards/rm_reward_func/std": 12.722545623779297, "step": 1159 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 492.0, "completions/mean_length": 351.09375, "completions/mean_terminated_length": 313.9615478515625, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.928, "grad_norm": 17.148536682128906, "kl": 0.40380859375, "learning_rate": 1e-06, "loss": 0.0482, "num_tokens": 15250796.0, "reward": 10.2752685546875, "reward_std": 4.209094047546387, "rewards/rm_reward_func/mean": 10.2752685546875, "rewards/rm_reward_func/std": 11.812329292297363, "step": 1160 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 407.0, "completions/max_terminated_length": 407.0, "completions/mean_length": 171.5, "completions/mean_terminated_length": 171.5, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "epoch": 0.9288, "grad_norm": 42.1457633972168, "kl": 1.83056640625, "learning_rate": 1e-06, "loss": 0.2306, "num_tokens": 15258540.0, "reward": 0.6884765625, "reward_std": 7.095283508300781, "rewards/rm_reward_func/mean": 0.6884765625, "rewards/rm_reward_func/std": 11.873205184936523, "step": 1161 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 369.46875, "completions/mean_terminated_length": 313.6956481933594, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9296, "grad_norm": 7.2761640548706055, "kl": 2.3427734375, "learning_rate": 1e-06, "loss": 0.0776, "num_tokens": 15272827.0, "reward": 9.095703125, "reward_std": 6.575770378112793, "rewards/rm_reward_func/mean": 9.095703125, "rewards/rm_reward_func/std": 15.32244873046875, "step": 1162 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 497.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 166.8125, "completions/mean_terminated_length": 166.8125, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "epoch": 0.9304, "grad_norm": 10.371278762817383, "kl": 0.509765625, "learning_rate": 1e-06, "loss": -0.0377, "num_tokens": 15280869.0, "reward": 5.901123046875, "reward_std": 2.3234264850616455, "rewards/rm_reward_func/mean": 5.901123046875, "rewards/rm_reward_func/std": 10.951936721801758, "step": 1163 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 296.21875, "completions/mean_terminated_length": 211.78260803222656, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9312, "grad_norm": 7.607293605804443, "kl": 2.140625, "learning_rate": 1e-06, "loss": -0.0091, "num_tokens": 15292900.0, "reward": -0.714080810546875, "reward_std": 6.104308128356934, "rewards/rm_reward_func/mean": -0.714080810546875, "rewards/rm_reward_func/std": 9.522336959838867, "step": 1164 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 332.71875, "completions/mean_terminated_length": 291.3461608886719, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.932, "grad_norm": 6.345226764678955, "kl": 1.32861328125, "learning_rate": 1e-06, "loss": 0.0391, "num_tokens": 15306323.0, "reward": 9.87939453125, "reward_std": 8.231972694396973, "rewards/rm_reward_func/mean": 9.87939453125, "rewards/rm_reward_func/std": 16.97296142578125, "step": 1165 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 296.71875, "completions/mean_terminated_length": 274.4482727050781, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9328, "grad_norm": 11.910536766052246, "kl": 3.41845703125, "learning_rate": 1e-06, "loss": -0.0196, "num_tokens": 15318378.0, "reward": -2.5654296875, "reward_std": 8.650764465332031, "rewards/rm_reward_func/mean": -2.5654296875, "rewards/rm_reward_func/std": 14.586143493652344, "step": 1166 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 370.0625, "completions/mean_terminated_length": 337.3077087402344, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9336, "grad_norm": 5.885568618774414, "kl": 0.9931640625, "learning_rate": 1e-06, "loss": 0.0021, "num_tokens": 15334220.0, "reward": 14.67633056640625, "reward_std": 8.415434837341309, "rewards/rm_reward_func/mean": 14.67633056640625, "rewards/rm_reward_func/std": 19.035816192626953, "step": 1167 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 492.0, "completions/mean_length": 397.21875, "completions/mean_terminated_length": 389.5666809082031, "completions/min_length": 167.0, "completions/min_terminated_length": 167.0, "epoch": 0.9344, "grad_norm": 11.35747241973877, "kl": 2.22021484375, "learning_rate": 1e-06, "loss": 0.0588, "num_tokens": 15349371.0, "reward": 7.377197265625, "reward_std": 6.683224678039551, "rewards/rm_reward_func/mean": 7.377197265625, "rewards/rm_reward_func/std": 13.624530792236328, "step": 1168 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 476.0, "completions/mean_length": 369.375, "completions/mean_terminated_length": 349.0000305175781, "completions/min_length": 157.0, "completions/min_terminated_length": 157.0, "epoch": 0.9352, "grad_norm": 28.526676177978516, "kl": 4.8515625, "learning_rate": 1e-06, "loss": 0.1177, "num_tokens": 15363063.0, "reward": 3.8070831298828125, "reward_std": 14.380706787109375, "rewards/rm_reward_func/mean": 3.8070831298828125, "rewards/rm_reward_func/std": 15.08162784576416, "step": 1169 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 500.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 277.78125, "completions/mean_terminated_length": 277.78125, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.936, "grad_norm": 7.368760585784912, "kl": 0.72314453125, "learning_rate": 1e-06, "loss": 0.0015, "num_tokens": 15373880.0, "reward": 2.238525390625, "reward_std": 2.461265802383423, "rewards/rm_reward_func/mean": 2.238525390625, "rewards/rm_reward_func/std": 6.371425628662109, "step": 1170 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 371.5, "completions/mean_terminated_length": 324.66668701171875, "completions/min_length": 94.0, "completions/min_terminated_length": 94.0, "epoch": 0.9368, "grad_norm": 23.937978744506836, "kl": 2.81396484375, "learning_rate": 1e-06, "loss": 0.0015, "num_tokens": 15387984.0, "reward": -3.2255859375, "reward_std": 8.331253051757812, "rewards/rm_reward_func/mean": -3.2255859375, "rewards/rm_reward_func/std": 10.779619216918945, "step": 1171 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 378.65625, "completions/mean_terminated_length": 318.04547119140625, "completions/min_length": 103.0, "completions/min_terminated_length": 103.0, "epoch": 0.9376, "grad_norm": 15.15499496459961, "kl": 4.59619140625, "learning_rate": 1e-06, "loss": 0.0643, "num_tokens": 15402765.0, "reward": -3.593994140625, "reward_std": 6.05653190612793, "rewards/rm_reward_func/mean": -3.593994140625, "rewards/rm_reward_func/std": 10.450895309448242, "step": 1172 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 506.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 241.40625, "completions/mean_terminated_length": 241.40625, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "epoch": 0.9384, "grad_norm": 10.372428894042969, "kl": 2.1416015625, "learning_rate": 1e-06, "loss": 0.0091, "num_tokens": 15417634.0, "reward": 6.179931640625, "reward_std": 7.2046685218811035, "rewards/rm_reward_func/mean": 6.179931640625, "rewards/rm_reward_func/std": 8.590556144714355, "step": 1173 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 276.90625, "completions/mean_terminated_length": 222.6538543701172, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9392, "grad_norm": 14.381688117980957, "kl": 1.92138671875, "learning_rate": 1e-06, "loss": 0.0851, "num_tokens": 15429623.0, "reward": -2.0927734375, "reward_std": 4.637411117553711, "rewards/rm_reward_func/mean": -2.0927734375, "rewards/rm_reward_func/std": 13.167999267578125, "step": 1174 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 280.53125, "completions/mean_terminated_length": 265.1000061035156, "completions/min_length": 122.0, "completions/min_terminated_length": 122.0, "epoch": 0.94, "grad_norm": 32.04814529418945, "kl": 7.716796875, "learning_rate": 1e-06, "loss": 0.1969, "num_tokens": 15440720.0, "reward": -1.5761604309082031, "reward_std": 14.394807815551758, "rewards/rm_reward_func/mean": -1.5761604309082031, "rewards/rm_reward_func/std": 17.306682586669922, "step": 1175 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 365.46875, "completions/mean_terminated_length": 331.65386962890625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "epoch": 0.9408, "grad_norm": 6.667887210845947, "kl": 1.29052734375, "learning_rate": 1e-06, "loss": -0.0015, "num_tokens": 15454551.0, "reward": 3.3128662109375, "reward_std": 7.290994644165039, "rewards/rm_reward_func/mean": 3.3128662109375, "rewards/rm_reward_func/std": 11.585569381713867, "step": 1176 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 372.25, "completions/mean_terminated_length": 317.5652160644531, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9416, "grad_norm": 6.514392375946045, "kl": 3.54052734375, "learning_rate": 1e-06, "loss": 0.1126, "num_tokens": 15469031.0, "reward": 15.22454833984375, "reward_std": 8.186134338378906, "rewards/rm_reward_func/mean": 15.22454833984375, "rewards/rm_reward_func/std": 20.967836380004883, "step": 1177 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 352.0, "completions/max_terminated_length": 352.0, "completions/mean_length": 128.28125, "completions/mean_terminated_length": 128.28125, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9424, "grad_norm": 57.319156646728516, "kl": 0.97265625, "learning_rate": 1e-06, "loss": 0.0197, "num_tokens": 15475856.0, "reward": 4.0428466796875, "reward_std": 1.0468058586120605, "rewards/rm_reward_func/mean": 4.0428466796875, "rewards/rm_reward_func/std": 8.96422004699707, "step": 1178 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 466.0, "completions/max_terminated_length": 466.0, "completions/mean_length": 257.375, "completions/mean_terminated_length": 257.375, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "epoch": 0.9432, "grad_norm": 37.18678283691406, "kl": 4.50732421875, "learning_rate": 1e-06, "loss": 0.1144, "num_tokens": 15490860.0, "reward": -0.7977094650268555, "reward_std": 9.967428207397461, "rewards/rm_reward_func/mean": -0.7977094650268555, "rewards/rm_reward_func/std": 12.24376106262207, "step": 1179 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 274.75, "completions/mean_terminated_length": 208.3199920654297, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.944, "grad_norm": 11.060836791992188, "kl": 2.35205078125, "learning_rate": 1e-06, "loss": 0.0886, "num_tokens": 15502980.0, "reward": 4.240234375, "reward_std": 2.008146047592163, "rewards/rm_reward_func/mean": 4.240234375, "rewards/rm_reward_func/std": 12.89011287689209, "step": 1180 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 487.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 224.71875, "completions/mean_terminated_length": 224.71875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "epoch": 0.9448, "grad_norm": 25.126611709594727, "kl": 3.6806640625, "learning_rate": 1e-06, "loss": 0.0654, "num_tokens": 15512067.0, "reward": -1.59136962890625, "reward_std": 6.693836212158203, "rewards/rm_reward_func/mean": -1.59136962890625, "rewards/rm_reward_func/std": 13.480401039123535, "step": 1181 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 310.40625, "completions/mean_terminated_length": 296.9666748046875, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9456, "grad_norm": 6.820868968963623, "kl": 2.0498046875, "learning_rate": 1e-06, "loss": 0.0904, "num_tokens": 15527256.0, "reward": 11.24267578125, "reward_std": 6.773592472076416, "rewards/rm_reward_func/mean": 11.24267578125, "rewards/rm_reward_func/std": 16.010009765625, "step": 1182 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 324.3125, "completions/mean_terminated_length": 271.7599792480469, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9464, "grad_norm": 5.817383289337158, "kl": 1.25830078125, "learning_rate": 1e-06, "loss": 0.0173, "num_tokens": 15540034.0, "reward": 12.9453125, "reward_std": 6.853506088256836, "rewards/rm_reward_func/mean": 12.9453125, "rewards/rm_reward_func/std": 11.050248146057129, "step": 1183 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.59375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 394.96875, "completions/mean_terminated_length": 223.92308044433594, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9472, "grad_norm": 11.657125473022461, "kl": 1.68115234375, "learning_rate": 1e-06, "loss": 0.0629, "num_tokens": 15556841.0, "reward": 1.7704925537109375, "reward_std": 4.207978248596191, "rewards/rm_reward_func/mean": 1.7704925537109375, "rewards/rm_reward_func/std": 6.741135120391846, "step": 1184 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 461.0, "completions/mean_length": 329.9375, "completions/mean_terminated_length": 303.9285888671875, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.948, "grad_norm": 26.10344696044922, "kl": 4.341796875, "learning_rate": 1e-06, "loss": 0.2526, "num_tokens": 15570263.0, "reward": 7.865570068359375, "reward_std": 6.319413185119629, "rewards/rm_reward_func/mean": 7.865570068359375, "rewards/rm_reward_func/std": 15.726142883300781, "step": 1185 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 461.0, "completions/mean_length": 335.1875, "completions/mean_terminated_length": 276.25, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9488, "grad_norm": 6.295200824737549, "kl": 0.4013671875, "learning_rate": 1e-06, "loss": 0.0074, "num_tokens": 15584085.0, "reward": 7.826171875, "reward_std": 1.844343900680542, "rewards/rm_reward_func/mean": 7.826171875, "rewards/rm_reward_func/std": 16.900344848632812, "step": 1186 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 414.0, "completions/mean_length": 311.96875, "completions/mean_terminated_length": 255.95999145507812, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9496, "grad_norm": 7.316650867462158, "kl": 0.916015625, "learning_rate": 1e-06, "loss": -0.0087, "num_tokens": 15598652.0, "reward": -1.8782958984375, "reward_std": 4.917675495147705, "rewards/rm_reward_func/mean": -1.8782958984375, "rewards/rm_reward_func/std": 8.429694175720215, "step": 1187 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 484.0, "completions/mean_length": 367.78125, "completions/mean_terminated_length": 358.16668701171875, "completions/min_length": 204.0, "completions/min_terminated_length": 204.0, "epoch": 0.9504, "grad_norm": 27.917741775512695, "kl": 1.7177734375, "learning_rate": 1e-06, "loss": 0.0232, "num_tokens": 15612501.0, "reward": 11.6102294921875, "reward_std": 8.347414016723633, "rewards/rm_reward_func/mean": 11.6102294921875, "rewards/rm_reward_func/std": 13.412553787231445, "step": 1188 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 443.46875, "completions/mean_terminated_length": 420.625, "completions/min_length": 184.0, "completions/min_terminated_length": 184.0, "epoch": 0.9512, "grad_norm": 19.72421646118164, "kl": 3.552734375, "learning_rate": 1e-06, "loss": 0.1278, "num_tokens": 15628892.0, "reward": 5.533203125, "reward_std": 10.962414741516113, "rewards/rm_reward_func/mean": 5.533203125, "rewards/rm_reward_func/std": 24.766939163208008, "step": 1189 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 414.0, "completions/mean_length": 296.40625, "completions/mean_terminated_length": 289.45159912109375, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.952, "grad_norm": 8.21200180053711, "kl": 1.095703125, "learning_rate": 1e-06, "loss": 0.0381, "num_tokens": 15642537.0, "reward": 6.253631591796875, "reward_std": 4.122445106506348, "rewards/rm_reward_func/mean": 6.253631591796875, "rewards/rm_reward_func/std": 6.488741874694824, "step": 1190 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 486.0, "completions/mean_length": 342.65625, "completions/mean_terminated_length": 331.3666687011719, "completions/min_length": 97.0, "completions/min_terminated_length": 97.0, "epoch": 0.9528, "grad_norm": 7.0987677574157715, "kl": 2.54150390625, "learning_rate": 1e-06, "loss": 0.0774, "num_tokens": 15657294.0, "reward": 4.48828125, "reward_std": 5.943986892700195, "rewards/rm_reward_func/mean": 4.48828125, "rewards/rm_reward_func/std": 18.430313110351562, "step": 1191 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 337.0625, "completions/mean_terminated_length": 296.69232177734375, "completions/min_length": 137.0, "completions/min_terminated_length": 137.0, "epoch": 0.9536, "grad_norm": 10.321151733398438, "kl": 0.74853515625, "learning_rate": 1e-06, "loss": 0.0207, "num_tokens": 15670624.0, "reward": 8.99795150756836, "reward_std": 6.07137393951416, "rewards/rm_reward_func/mean": 8.99795150756836, "rewards/rm_reward_func/std": 11.28697395324707, "step": 1192 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 406.46875, "completions/mean_terminated_length": 365.1739196777344, "completions/min_length": 156.0, "completions/min_terminated_length": 156.0, "epoch": 0.9544, "grad_norm": 75.35392761230469, "kl": 8.7998046875, "learning_rate": 1e-06, "loss": 0.3095, "num_tokens": 15688439.0, "reward": -8.990234375, "reward_std": 6.966933250427246, "rewards/rm_reward_func/mean": -8.990234375, "rewards/rm_reward_func/std": 18.05649757385254, "step": 1193 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 239.8125, "completions/mean_terminated_length": 211.65516662597656, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9552, "grad_norm": 15.314480781555176, "kl": 1.806640625, "learning_rate": 1e-06, "loss": 0.0248, "num_tokens": 15700577.0, "reward": 3.8753662109375, "reward_std": 4.899765491485596, "rewards/rm_reward_func/mean": 3.8753662109375, "rewards/rm_reward_func/std": 8.12557315826416, "step": 1194 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 362.15625, "completions/mean_terminated_length": 327.5769348144531, "completions/min_length": 149.0, "completions/min_terminated_length": 149.0, "epoch": 0.956, "grad_norm": 78.24580383300781, "kl": 8.9765625, "learning_rate": 1e-06, "loss": 0.2552, "num_tokens": 15714486.0, "reward": -6.42059326171875, "reward_std": 9.736968994140625, "rewards/rm_reward_func/mean": -6.42059326171875, "rewards/rm_reward_func/std": 14.20551586151123, "step": 1195 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.28125, "completions/max_length": 512.0, "completions/max_terminated_length": 397.0, "completions/mean_length": 280.46875, "completions/mean_terminated_length": 189.86956787109375, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9568, "grad_norm": 31.360624313354492, "kl": 3.84814453125, "learning_rate": 1e-06, "loss": 0.19, "num_tokens": 15726301.0, "reward": 1.2685317993164062, "reward_std": 4.518694877624512, "rewards/rm_reward_func/mean": 1.2685317993164062, "rewards/rm_reward_func/std": 8.930562973022461, "step": 1196 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 274.03125, "completions/mean_terminated_length": 240.0357208251953, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9576, "grad_norm": 7.44788122177124, "kl": 0.564453125, "learning_rate": 1e-06, "loss": 0.0323, "num_tokens": 15738126.0, "reward": 8.062255859375, "reward_std": 1.784210443496704, "rewards/rm_reward_func/mean": 8.062255859375, "rewards/rm_reward_func/std": 6.24770450592041, "step": 1197 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 430.0, "completions/max_terminated_length": 430.0, "completions/mean_length": 200.1875, "completions/mean_terminated_length": 200.1875, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9584, "grad_norm": 46.06965637207031, "kl": 6.3017578125, "learning_rate": 1e-06, "loss": 0.2246, "num_tokens": 15746660.0, "reward": -1.352294921875, "reward_std": 7.574354648590088, "rewards/rm_reward_func/mean": -1.352294921875, "rewards/rm_reward_func/std": 14.081011772155762, "step": 1198 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 303.0625, "completions/mean_terminated_length": 264.370361328125, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9592, "grad_norm": 13.53447437286377, "kl": 3.17578125, "learning_rate": 1e-06, "loss": 0.0633, "num_tokens": 15761318.0, "reward": 6.076904296875, "reward_std": 8.05277156829834, "rewards/rm_reward_func/mean": 6.076904296875, "rewards/rm_reward_func/std": 10.448375701904297, "step": 1199 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 372.71875, "completions/mean_terminated_length": 358.3103332519531, "completions/min_length": 108.0, "completions/min_terminated_length": 108.0, "epoch": 0.96, "grad_norm": 82.56756591796875, "kl": 6.08251953125, "learning_rate": 1e-06, "loss": 0.1814, "num_tokens": 15777317.0, "reward": 0.806884765625, "reward_std": 8.0141019821167, "rewards/rm_reward_func/mean": 0.806884765625, "rewards/rm_reward_func/std": 20.149206161499023, "step": 1200 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 134.0, "completions/mean_terminated_length": 80.0, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9608, "grad_norm": 80.49437713623047, "kl": 2.533203125, "learning_rate": 1e-06, "loss": 0.3195, "num_tokens": 15786189.0, "reward": 2.772796630859375, "reward_std": 2.7184181213378906, "rewards/rm_reward_func/mean": 2.772796630859375, "rewards/rm_reward_func/std": 8.184410095214844, "step": 1201 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 355.71875, "completions/mean_terminated_length": 326.77777099609375, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9616, "grad_norm": 22.120071411132812, "kl": 0.51708984375, "learning_rate": 1e-06, "loss": 0.0185, "num_tokens": 15802404.0, "reward": 19.8399658203125, "reward_std": 4.591503620147705, "rewards/rm_reward_func/mean": 19.8399658203125, "rewards/rm_reward_func/std": 21.975711822509766, "step": 1202 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 399.125, "completions/mean_terminated_length": 383.0000305175781, "completions/min_length": 156.0, "completions/min_terminated_length": 156.0, "epoch": 0.9624, "grad_norm": 13.387299537658691, "kl": 4.68603515625, "learning_rate": 1e-06, "loss": 0.2318, "num_tokens": 15818008.0, "reward": 5.8818359375, "reward_std": 6.727931976318359, "rewards/rm_reward_func/mean": 5.8818359375, "rewards/rm_reward_func/std": 23.548294067382812, "step": 1203 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 284.125, "completions/mean_terminated_length": 241.92593383789062, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9632, "grad_norm": 5.043156147003174, "kl": 2.41845703125, "learning_rate": 1e-06, "loss": 0.096, "num_tokens": 15830828.0, "reward": 2.5513916015625, "reward_std": 6.220531463623047, "rewards/rm_reward_func/mean": 2.5513916015625, "rewards/rm_reward_func/std": 13.680570602416992, "step": 1204 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 477.0, "completions/mean_length": 379.25, "completions/mean_terminated_length": 365.5172424316406, "completions/min_length": 177.0, "completions/min_terminated_length": 177.0, "epoch": 0.964, "grad_norm": 19.86156463623047, "kl": 4.646484375, "learning_rate": 1e-06, "loss": 0.19, "num_tokens": 15844940.0, "reward": 12.568359375, "reward_std": 12.276927947998047, "rewards/rm_reward_func/mean": 12.568359375, "rewards/rm_reward_func/std": 18.882339477539062, "step": 1205 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 472.0, "completions/mean_length": 244.0625, "completions/mean_terminated_length": 226.20001220703125, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "epoch": 0.9648, "grad_norm": 35.16064453125, "kl": 3.6748046875, "learning_rate": 1e-06, "loss": -0.0097, "num_tokens": 15854710.0, "reward": 4.755615234375, "reward_std": 10.025222778320312, "rewards/rm_reward_func/mean": 4.755615234375, "rewards/rm_reward_func/std": 17.34681510925293, "step": 1206 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 480.0, "completions/mean_length": 330.71875, "completions/mean_terminated_length": 297.1481628417969, "completions/min_length": 148.0, "completions/min_terminated_length": 148.0, "epoch": 0.9656, "grad_norm": 20.283857345581055, "kl": 4.10986328125, "learning_rate": 1e-06, "loss": 0.1873, "num_tokens": 15869749.0, "reward": -1.422607421875, "reward_std": 7.128761291503906, "rewards/rm_reward_func/mean": -1.422607421875, "rewards/rm_reward_func/std": 8.89996337890625, "step": 1207 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 461.0, "completions/max_terminated_length": 461.0, "completions/mean_length": 228.8125, "completions/mean_terminated_length": 228.8125, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9664, "grad_norm": 14.01342487335205, "kl": 1.6328125, "learning_rate": 1e-06, "loss": 0.0612, "num_tokens": 15879927.0, "reward": 5.265869140625, "reward_std": 2.066286087036133, "rewards/rm_reward_func/mean": 5.265869140625, "rewards/rm_reward_func/std": 7.653271675109863, "step": 1208 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 509.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 301.34375, "completions/mean_terminated_length": 301.34375, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9672, "grad_norm": 16.027812957763672, "kl": 2.18359375, "learning_rate": 1e-06, "loss": -0.007, "num_tokens": 15895018.0, "reward": 5.47796630859375, "reward_std": 9.107280731201172, "rewards/rm_reward_func/mean": 5.47796630859375, "rewards/rm_reward_func/std": 13.736273765563965, "step": 1209 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 452.0, "completions/mean_length": 366.75, "completions/mean_terminated_length": 326.0799865722656, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "epoch": 0.968, "grad_norm": 6.4775848388671875, "kl": 0.8310546875, "learning_rate": 1e-06, "loss": -0.1347, "num_tokens": 15909162.0, "reward": 4.66473388671875, "reward_std": 10.003836631774902, "rewards/rm_reward_func/mean": 4.66473388671875, "rewards/rm_reward_func/std": 11.560744285583496, "step": 1210 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 442.0, "completions/max_terminated_length": 442.0, "completions/mean_length": 250.40625, "completions/mean_terminated_length": 250.40625, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9688, "grad_norm": 9.306504249572754, "kl": 1.1865234375, "learning_rate": 1e-06, "loss": 0.0649, "num_tokens": 15920199.0, "reward": 11.21533203125, "reward_std": 4.61986780166626, "rewards/rm_reward_func/mean": 11.21533203125, "rewards/rm_reward_func/std": 11.726616859436035, "step": 1211 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 428.0, "completions/mean_length": 276.8125, "completions/mean_terminated_length": 243.21429443359375, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9696, "grad_norm": 29.228607177734375, "kl": 1.453125, "learning_rate": 1e-06, "loss": 0.0867, "num_tokens": 15932337.0, "reward": -3.7728271484375, "reward_std": 2.3056063652038574, "rewards/rm_reward_func/mean": -3.7728271484375, "rewards/rm_reward_func/std": 5.636190891265869, "step": 1212 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 476.0, "completions/mean_length": 242.71875, "completions/mean_terminated_length": 204.25001525878906, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "epoch": 0.9704, "grad_norm": 28.853511810302734, "kl": 2.189453125, "learning_rate": 1e-06, "loss": 0.0766, "num_tokens": 15942880.0, "reward": 4.451416015625, "reward_std": 4.814798355102539, "rewards/rm_reward_func/mean": 4.451416015625, "rewards/rm_reward_func/std": 7.7672343254089355, "step": 1213 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 307.4375, "completions/mean_terminated_length": 260.23077392578125, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "epoch": 0.9712, "grad_norm": 8.984874725341797, "kl": 0.439697265625, "learning_rate": 1e-06, "loss": 0.0709, "num_tokens": 15954926.0, "reward": 2.17816162109375, "reward_std": 5.897883415222168, "rewards/rm_reward_func/mean": 2.17816162109375, "rewards/rm_reward_func/std": 6.458736419677734, "step": 1214 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 421.0, "completions/max_terminated_length": 421.0, "completions/mean_length": 164.15625, "completions/mean_terminated_length": 164.15625, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "epoch": 0.972, "grad_norm": 15.127213478088379, "kl": 0.443115234375, "learning_rate": 1e-06, "loss": 0.0297, "num_tokens": 15965971.0, "reward": 8.07522964477539, "reward_std": 1.5865893363952637, "rewards/rm_reward_func/mean": 8.07522964477539, "rewards/rm_reward_func/std": 9.159050941467285, "step": 1215 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 255.9375, "completions/mean_terminated_length": 247.6774139404297, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "epoch": 0.9728, "grad_norm": 42.37367630004883, "kl": 2.08447265625, "learning_rate": 1e-06, "loss": 0.0359, "num_tokens": 15979769.0, "reward": 1.78619384765625, "reward_std": 2.9838078022003174, "rewards/rm_reward_func/mean": 1.78619384765625, "rewards/rm_reward_func/std": 11.067867279052734, "step": 1216 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.3125, "completions/max_length": 512.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 346.59375, "completions/mean_terminated_length": 271.4090881347656, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9736, "grad_norm": 9.83594799041748, "kl": 4.3447265625, "learning_rate": 1e-06, "loss": 0.1791, "num_tokens": 15996100.0, "reward": -1.2491302490234375, "reward_std": 3.7429025173187256, "rewards/rm_reward_func/mean": -1.2491302490234375, "rewards/rm_reward_func/std": 8.419785499572754, "step": 1217 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 353.59375, "completions/mean_terminated_length": 343.0333557128906, "completions/min_length": 100.0, "completions/min_terminated_length": 100.0, "epoch": 0.9744, "grad_norm": 10.141213417053223, "kl": 2.3017578125, "learning_rate": 1e-06, "loss": 0.0357, "num_tokens": 16009191.0, "reward": 3.6845703125, "reward_std": 9.749015808105469, "rewards/rm_reward_func/mean": 3.6845703125, "rewards/rm_reward_func/std": 15.83315658569336, "step": 1218 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 451.40625, "completions/mean_terminated_length": 419.66668701171875, "completions/min_length": 304.0, "completions/min_terminated_length": 304.0, "epoch": 0.9752, "grad_norm": 6.541053771972656, "kl": 0.66943359375, "learning_rate": 1e-06, "loss": 0.0511, "num_tokens": 16026156.0, "reward": 10.817626953125, "reward_std": 10.195655822753906, "rewards/rm_reward_func/mean": 10.817626953125, "rewards/rm_reward_func/std": 14.443761825561523, "step": 1219 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 509.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 308.6875, "completions/mean_terminated_length": 308.6875, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "epoch": 0.976, "grad_norm": 7.883213996887207, "kl": 0.72705078125, "learning_rate": 1e-06, "loss": -0.0421, "num_tokens": 16040322.0, "reward": 2.47705078125, "reward_std": 7.948477745056152, "rewards/rm_reward_func/mean": 2.47705078125, "rewards/rm_reward_func/std": 14.740900039672852, "step": 1220 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 231.6875, "completions/mean_terminated_length": 231.6875, "completions/min_length": 40.0, "completions/min_terminated_length": 40.0, "epoch": 0.9768, "grad_norm": 16.915742874145508, "kl": 1.87890625, "learning_rate": 1e-06, "loss": 0.0358, "num_tokens": 16050816.0, "reward": -4.7006988525390625, "reward_std": 6.505680561065674, "rewards/rm_reward_func/mean": -4.7006988525390625, "rewards/rm_reward_func/std": 9.603677749633789, "step": 1221 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 248.0, "completions/mean_terminated_length": 210.2857208251953, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9776, "grad_norm": 7.012750625610352, "kl": 0.8623046875, "learning_rate": 1e-06, "loss": 0.0782, "num_tokens": 16061744.0, "reward": 12.907470703125, "reward_std": 4.333094596862793, "rewards/rm_reward_func/mean": 12.907470703125, "rewards/rm_reward_func/std": 8.020941734313965, "step": 1222 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 401.0, "completions/max_terminated_length": 401.0, "completions/mean_length": 228.8125, "completions/mean_terminated_length": 228.8125, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9784, "grad_norm": 6.7300801277160645, "kl": 0.89794921875, "learning_rate": 1e-06, "loss": 0.0436, "num_tokens": 16072642.0, "reward": -0.694091796875, "reward_std": 3.24855637550354, "rewards/rm_reward_func/mean": -0.694091796875, "rewards/rm_reward_func/std": 10.3416166305542, "step": 1223 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 339.0, "completions/max_terminated_length": 339.0, "completions/mean_length": 175.375, "completions/mean_terminated_length": 175.375, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9792, "grad_norm": 5.84846305847168, "kl": 1.20361328125, "learning_rate": 1e-06, "loss": 0.0497, "num_tokens": 16080534.0, "reward": 4.77056884765625, "reward_std": 2.0563995838165283, "rewards/rm_reward_func/mean": 4.77056884765625, "rewards/rm_reward_func/std": 4.059889316558838, "step": 1224 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 481.0, "completions/mean_length": 284.625, "completions/mean_terminated_length": 269.4666748046875, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "epoch": 0.98, "grad_norm": 8.25778865814209, "kl": 2.14990234375, "learning_rate": 1e-06, "loss": 0.0486, "num_tokens": 16091746.0, "reward": 6.4002685546875, "reward_std": 7.458786964416504, "rewards/rm_reward_func/mean": 6.4002685546875, "rewards/rm_reward_func/std": 13.408005714416504, "step": 1225 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 341.875, "completions/mean_terminated_length": 294.239990234375, "completions/min_length": 131.0, "completions/min_terminated_length": 131.0, "epoch": 0.9808, "grad_norm": 10.492647171020508, "kl": 2.84521484375, "learning_rate": 1e-06, "loss": 0.0935, "num_tokens": 16105958.0, "reward": -4.466461181640625, "reward_std": 5.502716064453125, "rewards/rm_reward_func/mean": -4.466461181640625, "rewards/rm_reward_func/std": 6.831468105316162, "step": 1226 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.40625, "completions/max_length": 512.0, "completions/max_terminated_length": 480.0, "completions/mean_length": 310.53125, "completions/mean_terminated_length": 172.68421936035156, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9816, "grad_norm": 8.328527450561523, "kl": 3.4453125, "learning_rate": 1e-06, "loss": 0.1165, "num_tokens": 16120351.0, "reward": -7.10888671875, "reward_std": 3.2640366554260254, "rewards/rm_reward_func/mean": -7.10888671875, "rewards/rm_reward_func/std": 10.982377052307129, "step": 1227 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 439.0, "completions/max_terminated_length": 439.0, "completions/mean_length": 246.4375, "completions/mean_terminated_length": 246.4375, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9824, "grad_norm": 10.678609848022461, "kl": 3.5791015625, "learning_rate": 1e-06, "loss": 0.1131, "num_tokens": 16133981.0, "reward": 5.48870849609375, "reward_std": 6.148322105407715, "rewards/rm_reward_func/mean": 5.48870849609375, "rewards/rm_reward_func/std": 13.595304489135742, "step": 1228 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 470.0, "completions/max_terminated_length": 470.0, "completions/mean_length": 254.1875, "completions/mean_terminated_length": 254.1875, "completions/min_length": 97.0, "completions/min_terminated_length": 97.0, "epoch": 0.9832, "grad_norm": 7.430377006530762, "kl": 1.12890625, "learning_rate": 1e-06, "loss": -0.0441, "num_tokens": 16145083.0, "reward": 8.41162109375, "reward_std": 8.556034088134766, "rewards/rm_reward_func/mean": 8.41162109375, "rewards/rm_reward_func/std": 13.60086727142334, "step": 1229 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 477.0, "completions/mean_length": 232.53125, "completions/mean_terminated_length": 213.90000915527344, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.984, "grad_norm": 5.466613292694092, "kl": 1.316650390625, "learning_rate": 1e-06, "loss": -0.0194, "num_tokens": 16157052.0, "reward": 13.6875, "reward_std": 7.069519996643066, "rewards/rm_reward_func/mean": 13.6875, "rewards/rm_reward_func/std": 15.511878967285156, "step": 1230 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 479.0, "completions/mean_length": 256.96875, "completions/mean_terminated_length": 239.9666748046875, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "epoch": 0.9848, "grad_norm": 14.960919380187988, "kl": 1.33642578125, "learning_rate": 1e-06, "loss": 0.048, "num_tokens": 16167787.0, "reward": 13.341796875, "reward_std": 7.339491844177246, "rewards/rm_reward_func/mean": 13.341796875, "rewards/rm_reward_func/std": 19.08413314819336, "step": 1231 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 271.5625, "completions/mean_terminated_length": 263.80645751953125, "completions/min_length": 102.0, "completions/min_terminated_length": 102.0, "epoch": 0.9856, "grad_norm": 21.033065795898438, "kl": 3.0126953125, "learning_rate": 1e-06, "loss": 0.0807, "num_tokens": 16178821.0, "reward": 1.5001983642578125, "reward_std": 6.436107635498047, "rewards/rm_reward_func/mean": 1.5001983642578125, "rewards/rm_reward_func/std": 14.546778678894043, "step": 1232 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 342.78125, "completions/mean_terminated_length": 286.375, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "epoch": 0.9864, "grad_norm": 11.773747444152832, "kl": 6.388427734375, "learning_rate": 1e-06, "loss": 0.3401, "num_tokens": 16192638.0, "reward": -4.092529296875, "reward_std": 6.334899425506592, "rewards/rm_reward_func/mean": -4.092529296875, "rewards/rm_reward_func/std": 12.157096862792969, "step": 1233 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.5, "completions/max_length": 512.0, "completions/max_terminated_length": 492.0, "completions/mean_length": 448.21875, "completions/mean_terminated_length": 384.4375, "completions/min_length": 201.0, "completions/min_terminated_length": 201.0, "epoch": 0.9872, "grad_norm": 14.910449028015137, "kl": 4.833984375, "learning_rate": 1e-06, "loss": 0.1394, "num_tokens": 16210957.0, "reward": -4.65869140625, "reward_std": 6.343834400177002, "rewards/rm_reward_func/mean": -4.65869140625, "rewards/rm_reward_func/std": 21.47957420349121, "step": 1234 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 258.75, "completions/mean_terminated_length": 187.83999633789062, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "epoch": 0.988, "grad_norm": 31.964218139648438, "kl": 1.8095703125, "learning_rate": 1e-06, "loss": 0.0506, "num_tokens": 16222165.0, "reward": -6.220703125, "reward_std": 4.152710914611816, "rewards/rm_reward_func/mean": -6.220703125, "rewards/rm_reward_func/std": 10.58847427368164, "step": 1235 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 472.0, "completions/max_terminated_length": 472.0, "completions/mean_length": 198.71875, "completions/mean_terminated_length": 198.71875, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "epoch": 0.9888, "grad_norm": 5.49161958694458, "kl": 1.7666015625, "learning_rate": 1e-06, "loss": -0.0291, "num_tokens": 16231932.0, "reward": -1.51971435546875, "reward_std": 2.968614101409912, "rewards/rm_reward_func/mean": -1.51971435546875, "rewards/rm_reward_func/std": 9.40853500366211, "step": 1236 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 256.84375, "completions/mean_terminated_length": 239.83334350585938, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9896, "grad_norm": 7.274431228637695, "kl": 0.58251953125, "learning_rate": 1e-06, "loss": -0.0121, "num_tokens": 16243055.0, "reward": 13.570831298828125, "reward_std": 3.243131637573242, "rewards/rm_reward_func/mean": 13.570831298828125, "rewards/rm_reward_func/std": 12.231443405151367, "step": 1237 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 423.1875, "completions/mean_terminated_length": 369.8999938964844, "completions/min_length": 179.0, "completions/min_terminated_length": 179.0, "epoch": 0.9904, "grad_norm": 8.700695037841797, "kl": 1.9287109375, "learning_rate": 1e-06, "loss": 0.0465, "num_tokens": 16259893.0, "reward": -7.0559539794921875, "reward_std": 5.43247652053833, "rewards/rm_reward_func/mean": -7.0559539794921875, "rewards/rm_reward_func/std": 7.467658519744873, "step": 1238 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 320.25, "completions/mean_terminated_length": 284.7407531738281, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9912, "grad_norm": 14.817831039428711, "kl": 1.71875, "learning_rate": 1e-06, "loss": 0.0421, "num_tokens": 16273869.0, "reward": 7.86328125, "reward_std": 7.426191806793213, "rewards/rm_reward_func/mean": 7.86328125, "rewards/rm_reward_func/std": 9.81981372833252, "step": 1239 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.34375, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 327.84375, "completions/mean_terminated_length": 231.38095092773438, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.992, "grad_norm": 10.064021110534668, "kl": 1.5625, "learning_rate": 1e-06, "loss": 0.0699, "num_tokens": 16288208.0, "reward": 5.3447265625, "reward_std": 4.9022955894470215, "rewards/rm_reward_func/mean": 5.3447265625, "rewards/rm_reward_func/std": 9.280166625976562, "step": 1240 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 480.0, "completions/max_terminated_length": 480.0, "completions/mean_length": 348.71875, "completions/mean_terminated_length": 348.71875, "completions/min_length": 184.0, "completions/min_terminated_length": 184.0, "epoch": 0.9928, "grad_norm": 24.340036392211914, "kl": 0.641357421875, "learning_rate": 1e-06, "loss": 0.0044, "num_tokens": 16301511.0, "reward": 2.7781982421875, "reward_std": 5.280330181121826, "rewards/rm_reward_func/mean": 2.7781982421875, "rewards/rm_reward_func/std": 7.673682689666748, "step": 1241 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 320.46875, "completions/mean_terminated_length": 293.1071472167969, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "epoch": 0.9936, "grad_norm": 5.585651397705078, "kl": 2.2373046875, "learning_rate": 1e-06, "loss": 0.0234, "num_tokens": 16314094.0, "reward": 5.1142578125, "reward_std": 4.644550323486328, "rewards/rm_reward_func/mean": 5.1142578125, "rewards/rm_reward_func/std": 6.202043056488037, "step": 1242 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 260.65625, "completions/mean_terminated_length": 202.6538543701172, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "epoch": 0.9944, "grad_norm": 38.17617416381836, "kl": 7.8544921875, "learning_rate": 1e-06, "loss": 0.3338, "num_tokens": 16327371.0, "reward": -8.823577880859375, "reward_std": 5.635340213775635, "rewards/rm_reward_func/mean": -8.823577880859375, "rewards/rm_reward_func/std": 15.1368989944458, "step": 1243 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 302.96875, "completions/mean_terminated_length": 281.3448181152344, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "epoch": 0.9952, "grad_norm": 13.918490409851074, "kl": 5.79296875, "learning_rate": 1e-06, "loss": 0.1535, "num_tokens": 16340482.0, "reward": -7.033935546875, "reward_std": 8.026933670043945, "rewards/rm_reward_func/mean": -7.033935546875, "rewards/rm_reward_func/std": 13.346774101257324, "step": 1244 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 217.5625, "completions/mean_terminated_length": 149.61538696289062, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "epoch": 0.996, "grad_norm": 5.955099105834961, "kl": 1.0595703125, "learning_rate": 1e-06, "loss": -0.0099, "num_tokens": 16350628.0, "reward": 6.19781494140625, "reward_std": 5.490054130554199, "rewards/rm_reward_func/mean": 6.19781494140625, "rewards/rm_reward_func/std": 7.488785266876221, "step": 1245 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 479.0, "completions/mean_length": 313.625, "completions/mean_terminated_length": 293.10345458984375, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9968, "grad_norm": 16.472368240356445, "kl": 6.9775390625, "learning_rate": 1e-06, "loss": 0.2136, "num_tokens": 16364000.0, "reward": -2.907135009765625, "reward_std": 5.292792320251465, "rewards/rm_reward_func/mean": -2.907135009765625, "rewards/rm_reward_func/std": 9.868986129760742, "step": 1246 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 282.6875, "completions/mean_terminated_length": 275.2903137207031, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "epoch": 0.9976, "grad_norm": 41.74302673339844, "kl": 11.40576171875, "learning_rate": 1e-06, "loss": 0.2963, "num_tokens": 16375222.0, "reward": -6.39111328125, "reward_std": 3.2877705097198486, "rewards/rm_reward_func/mean": -6.39111328125, "rewards/rm_reward_func/std": 19.594057083129883, "step": 1247 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 473.0, "completions/mean_length": 304.3125, "completions/mean_terminated_length": 235.08334350585938, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "epoch": 0.9984, "grad_norm": 23.52429962158203, "kl": 5.68408203125, "learning_rate": 1e-06, "loss": -0.0276, "num_tokens": 16387192.0, "reward": -7.197265625, "reward_std": 10.437725067138672, "rewards/rm_reward_func/mean": -7.197265625, "rewards/rm_reward_func/std": 14.508234024047852, "step": 1248 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 441.0, "completions/max_terminated_length": 441.0, "completions/mean_length": 246.1875, "completions/mean_terminated_length": 246.1875, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "epoch": 0.9992, "grad_norm": 11.459190368652344, "kl": 6.080078125, "learning_rate": 1e-06, "loss": 0.1161, "num_tokens": 16397814.0, "reward": -1.294921875, "reward_std": 7.976162433624268, "rewards/rm_reward_func/mean": -1.294921875, "rewards/rm_reward_func/std": 13.71561336517334, "step": 1249 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 485.0, "completions/max_terminated_length": 485.0, "completions/mean_length": 347.75, "completions/mean_terminated_length": 347.75, "completions/min_length": 136.0, "completions/min_terminated_length": 136.0, "epoch": 1.0, "grad_norm": 22.80115509033203, "kl": 4.734375, "learning_rate": 1e-06, "loss": 0.0654, "num_tokens": 16412325.0, "reward": -4.2939453125, "reward_std": 7.761114120483398, "rewards/rm_reward_func/mean": -4.2939453125, "rewards/rm_reward_func/std": 18.258264541625977, "step": 1250 } ], "logging_steps": 1, "max_steps": 1250, "num_input_tokens_seen": 16412325, "num_train_epochs": 1, "save_steps": 250, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 0.0, "train_batch_size": 8, "trial_name": null, "trial_params": null }