Files
Llama-3.1-8B-RM8B/trainer_state.json
ModelHub XC bf279aea9e 初始化项目,由ModelHub XC社区提供模型
Model: yapeichang/Llama-3.1-8B-RM8B
Source: Original Platform
2026-07-28 18:38:09 +08:00

31285 lines
1.1 MiB

{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 1.0,
"eval_steps": 500,
"global_step": 1250,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 470.0,
"completions/mean_length": 276.6875,
"completions/mean_terminated_length": 184.60870361328125,
"completions/min_length": 30.0,
"completions/min_terminated_length": 30.0,
"epoch": 0.0008,
"grad_norm": 3.820082187652588,
"kl": 0.0003216266632080078,
"learning_rate": 0.0,
"loss": -0.1909,
"num_tokens": 15006.0,
"reward": -8.20977783203125,
"reward_std": 4.4750261306762695,
"rewards/rm_reward_func/mean": -8.20977783203125,
"rewards/rm_reward_func/std": 7.660107135772705,
"step": 1
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.4375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 424.0,
"completions/mean_length": 313.625,
"completions/mean_terminated_length": 159.3333282470703,
"completions/min_length": 9.0,
"completions/min_terminated_length": 9.0,
"epoch": 0.0016,
"grad_norm": 4.830166339874268,
"kl": 0.0002695322036743164,
"learning_rate": 1.5873015873015872e-08,
"loss": 0.0694,
"num_tokens": 28018.0,
"reward": -11.44384765625,
"reward_std": 2.6692183017730713,
"rewards/rm_reward_func/mean": -11.44384765625,
"rewards/rm_reward_func/std": 6.70808744430542,
"step": 2
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.40625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 501.0,
"completions/mean_length": 374.28125,
"completions/mean_terminated_length": 280.0526428222656,
"completions/min_length": 83.0,
"completions/min_terminated_length": 83.0,
"epoch": 0.0024,
"grad_norm": 2.5839879512786865,
"kl": 0.0003209114074707031,
"learning_rate": 3.1746031746031744e-08,
"loss": -0.1093,
"num_tokens": 42787.0,
"reward": -13.593017578125,
"reward_std": 5.063211917877197,
"rewards/rm_reward_func/mean": -13.593017578125,
"rewards/rm_reward_func/std": 8.933624267578125,
"step": 3
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 467.0,
"completions/mean_length": 176.3125,
"completions/mean_terminated_length": 141.58621215820312,
"completions/min_length": 85.0,
"completions/min_terminated_length": 85.0,
"epoch": 0.0032,
"grad_norm": 3.8852806091308594,
"kl": 0.00033855438232421875,
"learning_rate": 4.7619047619047613e-08,
"loss": -0.0693,
"num_tokens": 50509.0,
"reward": -8.043212890625,
"reward_std": 6.7915544509887695,
"rewards/rm_reward_func/mean": -8.043212890625,
"rewards/rm_reward_func/std": 9.116313934326172,
"step": 4
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 450.0,
"completions/mean_length": 240.59375,
"completions/mean_terminated_length": 212.51724243164062,
"completions/min_length": 73.0,
"completions/min_terminated_length": 73.0,
"epoch": 0.004,
"grad_norm": 2.737971544265747,
"kl": 0.0002636909484863281,
"learning_rate": 6.349206349206349e-08,
"loss": -0.0144,
"num_tokens": 61880.0,
"reward": -7.558570861816406,
"reward_std": 4.206568717956543,
"rewards/rm_reward_func/mean": -7.558570861816406,
"rewards/rm_reward_func/std": 6.464176177978516,
"step": 5
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 449.0,
"completions/mean_length": 268.5,
"completions/mean_terminated_length": 187.33334350585938,
"completions/min_length": 38.0,
"completions/min_terminated_length": 38.0,
"epoch": 0.0048,
"grad_norm": 2.704625368118286,
"kl": 0.00022745132446289062,
"learning_rate": 7.936507936507936e-08,
"loss": -0.0148,
"num_tokens": 73048.0,
"reward": -8.5592041015625,
"reward_std": 6.456365585327148,
"rewards/rm_reward_func/mean": -8.5592041015625,
"rewards/rm_reward_func/std": 6.686816692352295,
"step": 6
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 308.0,
"completions/mean_length": 255.34375,
"completions/mean_terminated_length": 169.7916717529297,
"completions/min_length": 24.0,
"completions/min_terminated_length": 24.0,
"epoch": 0.0056,
"grad_norm": 3.0464067459106445,
"kl": 0.00023818016052246094,
"learning_rate": 9.523809523809523e-08,
"loss": 0.2187,
"num_tokens": 84043.0,
"reward": -10.86029052734375,
"reward_std": 5.81391716003418,
"rewards/rm_reward_func/mean": -10.86029052734375,
"rewards/rm_reward_func/std": 7.67104959487915,
"step": 7
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.46875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 498.0,
"completions/mean_length": 370.15625,
"completions/mean_terminated_length": 245.0,
"completions/min_length": 61.0,
"completions/min_terminated_length": 61.0,
"epoch": 0.0064,
"grad_norm": 2.1677191257476807,
"kl": 0.0003154277801513672,
"learning_rate": 1.111111111111111e-07,
"loss": -0.0946,
"num_tokens": 97960.0,
"reward": -9.22998046875,
"reward_std": 4.78300666809082,
"rewards/rm_reward_func/mean": -9.22998046875,
"rewards/rm_reward_func/std": 15.388681411743164,
"step": 8
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 419.0,
"completions/mean_length": 355.5,
"completions/mean_terminated_length": 294.2608642578125,
"completions/min_length": 181.0,
"completions/min_terminated_length": 181.0,
"epoch": 0.0072,
"grad_norm": 2.2008018493652344,
"kl": 0.00035500526428222656,
"learning_rate": 1.2698412698412698e-07,
"loss": 0.0485,
"num_tokens": 113944.0,
"reward": -7.944580078125,
"reward_std": 4.6264190673828125,
"rewards/rm_reward_func/mean": -7.944580078125,
"rewards/rm_reward_func/std": 7.673216342926025,
"step": 9
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.5,
"completions/max_length": 512.0,
"completions/max_terminated_length": 507.0,
"completions/mean_length": 398.65625,
"completions/mean_terminated_length": 285.3125,
"completions/min_length": 135.0,
"completions/min_terminated_length": 135.0,
"epoch": 0.008,
"grad_norm": 2.5219130516052246,
"kl": 0.0003647804260253906,
"learning_rate": 1.4285714285714285e-07,
"loss": -0.0682,
"num_tokens": 130357.0,
"reward": -14.8125,
"reward_std": 3.55131196975708,
"rewards/rm_reward_func/mean": -14.8125,
"rewards/rm_reward_func/std": 6.269670486450195,
"step": 10
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.75,
"completions/max_length": 512.0,
"completions/max_terminated_length": 468.0,
"completions/mean_length": 438.03125,
"completions/mean_terminated_length": 216.125,
"completions/min_length": 37.0,
"completions/min_terminated_length": 37.0,
"epoch": 0.0088,
"grad_norm": 2.1736643314361572,
"kl": 0.00029921531677246094,
"learning_rate": 1.5873015873015872e-07,
"loss": 0.2198,
"num_tokens": 148190.0,
"reward": -13.016815185546875,
"reward_std": 7.495743274688721,
"rewards/rm_reward_func/mean": -13.016815185546875,
"rewards/rm_reward_func/std": 8.30325698852539,
"step": 11
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 356.0,
"completions/mean_length": 178.5625,
"completions/mean_terminated_length": 144.0689697265625,
"completions/min_length": 25.0,
"completions/min_terminated_length": 25.0,
"epoch": 0.0096,
"grad_norm": 4.596688747406006,
"kl": 0.00018262863159179688,
"learning_rate": 1.7460317460317458e-07,
"loss": 0.0833,
"num_tokens": 156104.0,
"reward": 0.974029541015625,
"reward_std": 4.125612735748291,
"rewards/rm_reward_func/mean": 0.974029541015625,
"rewards/rm_reward_func/std": 9.185401916503906,
"step": 12
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 404.0,
"completions/max_terminated_length": 404.0,
"completions/mean_length": 250.25,
"completions/mean_terminated_length": 250.25,
"completions/min_length": 33.0,
"completions/min_terminated_length": 33.0,
"epoch": 0.0104,
"grad_norm": 3.2415318489074707,
"kl": 0.0003223419189453125,
"learning_rate": 1.9047619047619045e-07,
"loss": -0.1464,
"num_tokens": 166208.0,
"reward": -8.69970703125,
"reward_std": 4.659298419952393,
"rewards/rm_reward_func/mean": -8.69970703125,
"rewards/rm_reward_func/std": 5.567142963409424,
"step": 13
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 487.0,
"completions/mean_length": 328.03125,
"completions/mean_terminated_length": 244.4091033935547,
"completions/min_length": 29.0,
"completions/min_terminated_length": 29.0,
"epoch": 0.0112,
"grad_norm": 2.257594347000122,
"kl": 0.0003323554992675781,
"learning_rate": 2.0634920634920632e-07,
"loss": -0.0784,
"num_tokens": 178833.0,
"reward": -8.506607055664062,
"reward_std": 7.695546627044678,
"rewards/rm_reward_func/mean": -8.506607055664062,
"rewards/rm_reward_func/std": 8.626959800720215,
"step": 14
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 274.65625,
"completions/mean_terminated_length": 150.33334350585938,
"completions/min_length": 28.0,
"completions/min_terminated_length": 28.0,
"epoch": 0.012,
"grad_norm": 2.8894879817962646,
"kl": 0.00014698505401611328,
"learning_rate": 2.222222222222222e-07,
"loss": 0.1296,
"num_tokens": 192198.0,
"reward": -8.6026611328125,
"reward_std": 7.466651916503906,
"rewards/rm_reward_func/mean": -8.6026611328125,
"rewards/rm_reward_func/std": 8.928447723388672,
"step": 15
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 431.0,
"completions/max_terminated_length": 431.0,
"completions/mean_length": 163.53125,
"completions/mean_terminated_length": 163.53125,
"completions/min_length": 29.0,
"completions/min_terminated_length": 29.0,
"epoch": 0.0128,
"grad_norm": 6.473876476287842,
"kl": 0.00042629241943359375,
"learning_rate": 2.3809523809523806e-07,
"loss": -0.1249,
"num_tokens": 203023.0,
"reward": -5.33013916015625,
"reward_std": 5.538287162780762,
"rewards/rm_reward_func/mean": -5.33013916015625,
"rewards/rm_reward_func/std": 8.5722074508667,
"step": 16
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 499.0,
"completions/mean_length": 325.0625,
"completions/mean_terminated_length": 272.7200012207031,
"completions/min_length": 27.0,
"completions/min_terminated_length": 27.0,
"epoch": 0.0136,
"grad_norm": 2.029867172241211,
"kl": 0.0003304481506347656,
"learning_rate": 2.5396825396825396e-07,
"loss": 0.0048,
"num_tokens": 215681.0,
"reward": -8.299209594726562,
"reward_std": 5.960148811340332,
"rewards/rm_reward_func/mean": -8.299209594726562,
"rewards/rm_reward_func/std": 9.681968688964844,
"step": 17
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 467.0,
"completions/mean_length": 309.46875,
"completions/mean_terminated_length": 217.4091033935547,
"completions/min_length": 15.0,
"completions/min_terminated_length": 15.0,
"epoch": 0.0144,
"grad_norm": 3.225783348083496,
"kl": 0.0002518892288208008,
"learning_rate": 2.698412698412698e-07,
"loss": 0.454,
"num_tokens": 228784.0,
"reward": -12.326904296875,
"reward_std": 6.084526062011719,
"rewards/rm_reward_func/mean": -12.326904296875,
"rewards/rm_reward_func/std": 10.45023250579834,
"step": 18
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 496.0,
"completions/mean_length": 246.84375,
"completions/mean_terminated_length": 238.29031372070312,
"completions/min_length": 66.0,
"completions/min_terminated_length": 66.0,
"epoch": 0.0152,
"grad_norm": 3.1062114238739014,
"kl": 0.00031566619873046875,
"learning_rate": 2.857142857142857e-07,
"loss": -0.0427,
"num_tokens": 238891.0,
"reward": -5.867828369140625,
"reward_std": 5.073713302612305,
"rewards/rm_reward_func/mean": -5.867828369140625,
"rewards/rm_reward_func/std": 6.714073181152344,
"step": 19
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 463.0,
"completions/mean_length": 279.0625,
"completions/mean_terminated_length": 271.5483703613281,
"completions/min_length": 19.0,
"completions/min_terminated_length": 19.0,
"epoch": 0.016,
"grad_norm": 2.1619873046875,
"kl": 0.00023698806762695312,
"learning_rate": 3.0158730158730156e-07,
"loss": -0.1689,
"num_tokens": 249645.0,
"reward": -3.2251663208007812,
"reward_std": 4.3877034187316895,
"rewards/rm_reward_func/mean": -3.2251663208007812,
"rewards/rm_reward_func/std": 5.818446159362793,
"step": 20
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 490.0,
"completions/mean_length": 282.40625,
"completions/mean_terminated_length": 258.6551818847656,
"completions/min_length": 31.0,
"completions/min_terminated_length": 31.0,
"epoch": 0.0168,
"grad_norm": 4.3518900871276855,
"kl": 0.0003197193145751953,
"learning_rate": 3.1746031746031743e-07,
"loss": 0.1906,
"num_tokens": 261154.0,
"reward": -8.756103515625,
"reward_std": 3.641651153564453,
"rewards/rm_reward_func/mean": -8.756103515625,
"rewards/rm_reward_func/std": 4.865791320800781,
"step": 21
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 256.125,
"completions/mean_terminated_length": 184.47999572753906,
"completions/min_length": 25.0,
"completions/min_terminated_length": 25.0,
"epoch": 0.0176,
"grad_norm": 3.306163787841797,
"kl": 0.00031304359436035156,
"learning_rate": 3.333333333333333e-07,
"loss": 0.1021,
"num_tokens": 272150.0,
"reward": -13.5908203125,
"reward_std": 7.745641708374023,
"rewards/rm_reward_func/mean": -13.5908203125,
"rewards/rm_reward_func/std": 8.889524459838867,
"step": 22
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 436.0,
"completions/mean_length": 355.59375,
"completions/mean_terminated_length": 261.75,
"completions/min_length": 18.0,
"completions/min_terminated_length": 18.0,
"epoch": 0.0184,
"grad_norm": 2.356691837310791,
"kl": 0.0002713203430175781,
"learning_rate": 3.4920634920634917e-07,
"loss": -0.1426,
"num_tokens": 286265.0,
"reward": -8.7578125,
"reward_std": 4.64816951751709,
"rewards/rm_reward_func/mean": -8.7578125,
"rewards/rm_reward_func/std": 10.985777854919434,
"step": 23
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 238.65625,
"completions/mean_terminated_length": 147.5416717529297,
"completions/min_length": 18.0,
"completions/min_terminated_length": 18.0,
"epoch": 0.0192,
"grad_norm": 5.42078161239624,
"kl": 0.0004291534423828125,
"learning_rate": 3.6507936507936504e-07,
"loss": 0.0164,
"num_tokens": 297494.0,
"reward": -6.644187927246094,
"reward_std": 5.260914325714111,
"rewards/rm_reward_func/mean": -6.644187927246094,
"rewards/rm_reward_func/std": 10.8004150390625,
"step": 24
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 481.0,
"completions/mean_length": 231.4375,
"completions/mean_terminated_length": 202.41378784179688,
"completions/min_length": 34.0,
"completions/min_terminated_length": 34.0,
"epoch": 0.02,
"grad_norm": 5.109860420227051,
"kl": 0.00030303001403808594,
"learning_rate": 3.809523809523809e-07,
"loss": 0.084,
"num_tokens": 309052.0,
"reward": -5.954833984375,
"reward_std": 5.726484298706055,
"rewards/rm_reward_func/mean": -5.954833984375,
"rewards/rm_reward_func/std": 7.126110553741455,
"step": 25
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 193.0,
"completions/mean_length": 240.5625,
"completions/mean_terminated_length": 77.70000457763672,
"completions/min_length": 35.0,
"completions/min_terminated_length": 35.0,
"epoch": 0.0208,
"grad_norm": 3.4662835597991943,
"kl": 0.00029218196868896484,
"learning_rate": 3.968253968253968e-07,
"loss": 0.1907,
"num_tokens": 320334.0,
"reward": -14.75677490234375,
"reward_std": 4.248941421508789,
"rewards/rm_reward_func/mean": -14.75677490234375,
"rewards/rm_reward_func/std": 7.078120708465576,
"step": 26
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.53125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 492.0,
"completions/mean_length": 450.09375,
"completions/mean_terminated_length": 379.933349609375,
"completions/min_length": 212.0,
"completions/min_terminated_length": 212.0,
"epoch": 0.0216,
"grad_norm": 1.6936448812484741,
"kl": 0.0002624988555908203,
"learning_rate": 4.1269841269841265e-07,
"loss": -0.0219,
"num_tokens": 340865.0,
"reward": -8.20947265625,
"reward_std": 6.251974582672119,
"rewards/rm_reward_func/mean": -8.20947265625,
"rewards/rm_reward_func/std": 6.31874942779541,
"step": 27
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 394.0,
"completions/max_terminated_length": 394.0,
"completions/mean_length": 167.78125,
"completions/mean_terminated_length": 167.78125,
"completions/min_length": 45.0,
"completions/min_terminated_length": 45.0,
"epoch": 0.0224,
"grad_norm": 4.310393333435059,
"kl": 0.0003063678741455078,
"learning_rate": 4.285714285714285e-07,
"loss": 0.0078,
"num_tokens": 350994.0,
"reward": -4.7073516845703125,
"reward_std": 3.5501272678375244,
"rewards/rm_reward_func/mean": -4.7073516845703125,
"rewards/rm_reward_func/std": 11.039022445678711,
"step": 28
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 493.0,
"completions/mean_length": 316.75,
"completions/mean_terminated_length": 199.60000610351562,
"completions/min_length": 26.0,
"completions/min_terminated_length": 26.0,
"epoch": 0.0232,
"grad_norm": 2.857746124267578,
"kl": 0.0002956390380859375,
"learning_rate": 4.444444444444444e-07,
"loss": 0.0622,
"num_tokens": 368186.0,
"reward": -8.3876953125,
"reward_std": 4.28533935546875,
"rewards/rm_reward_func/mean": -8.3876953125,
"rewards/rm_reward_func/std": 6.924084663391113,
"step": 29
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 453.0,
"completions/mean_length": 275.71875,
"completions/mean_terminated_length": 221.19232177734375,
"completions/min_length": 45.0,
"completions/min_terminated_length": 45.0,
"epoch": 0.024,
"grad_norm": 2.670747995376587,
"kl": 0.0003662109375,
"learning_rate": 4.6031746031746025e-07,
"loss": 0.0619,
"num_tokens": 379457.0,
"reward": -8.928466796875,
"reward_std": 8.38475227355957,
"rewards/rm_reward_func/mean": -8.928466796875,
"rewards/rm_reward_func/std": 13.34780502319336,
"step": 30
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.40625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 470.0,
"completions/mean_length": 298.5,
"completions/mean_terminated_length": 152.42105102539062,
"completions/min_length": 36.0,
"completions/min_terminated_length": 36.0,
"epoch": 0.0248,
"grad_norm": 2.7848010063171387,
"kl": 0.00027370452880859375,
"learning_rate": 4.761904761904761e-07,
"loss": 0.1278,
"num_tokens": 392985.0,
"reward": -14.843505859375,
"reward_std": 5.917230129241943,
"rewards/rm_reward_func/mean": -14.843505859375,
"rewards/rm_reward_func/std": 8.529240608215332,
"step": 31
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 488.0,
"completions/mean_length": 258.9375,
"completions/mean_terminated_length": 174.58334350585938,
"completions/min_length": 17.0,
"completions/min_terminated_length": 17.0,
"epoch": 0.0256,
"grad_norm": 4.443420886993408,
"kl": 0.0003578662872314453,
"learning_rate": 4.92063492063492e-07,
"loss": 0.128,
"num_tokens": 405471.0,
"reward": -4.92132568359375,
"reward_std": 4.653083801269531,
"rewards/rm_reward_func/mean": -4.92132568359375,
"rewards/rm_reward_func/std": 13.973512649536133,
"step": 32
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 463.0,
"completions/mean_length": 181.9375,
"completions/mean_terminated_length": 159.93333435058594,
"completions/min_length": 19.0,
"completions/min_terminated_length": 19.0,
"epoch": 0.0264,
"grad_norm": 4.213375091552734,
"kl": 0.0003600120544433594,
"learning_rate": 5.079365079365079e-07,
"loss": -0.1714,
"num_tokens": 415629.0,
"reward": -11.66253662109375,
"reward_std": 6.659981727600098,
"rewards/rm_reward_func/mean": -11.66253662109375,
"rewards/rm_reward_func/std": 6.839272975921631,
"step": 33
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 435.0,
"completions/mean_length": 236.0625,
"completions/mean_terminated_length": 158.8000030517578,
"completions/min_length": 33.0,
"completions/min_terminated_length": 33.0,
"epoch": 0.0272,
"grad_norm": 3.4807815551757812,
"kl": 0.00034689903259277344,
"learning_rate": 5.238095238095238e-07,
"loss": -0.0034,
"num_tokens": 429039.0,
"reward": -8.47998046875,
"reward_std": 5.434866905212402,
"rewards/rm_reward_func/mean": -8.47998046875,
"rewards/rm_reward_func/std": 7.376414775848389,
"step": 34
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 493.0,
"completions/mean_length": 336.46875,
"completions/mean_terminated_length": 244.52381896972656,
"completions/min_length": 73.0,
"completions/min_terminated_length": 73.0,
"epoch": 0.028,
"grad_norm": 2.5712099075317383,
"kl": 0.0003306865692138672,
"learning_rate": 5.396825396825396e-07,
"loss": 0.1245,
"num_tokens": 443430.0,
"reward": -7.155723571777344,
"reward_std": 6.052016735076904,
"rewards/rm_reward_func/mean": -7.155723571777344,
"rewards/rm_reward_func/std": 8.14072322845459,
"step": 35
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.40625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 230.0,
"completions/mean_length": 251.21875,
"completions/mean_terminated_length": 72.78947448730469,
"completions/min_length": 29.0,
"completions/min_terminated_length": 29.0,
"epoch": 0.0288,
"grad_norm": 3.459679126739502,
"kl": 0.0003533363342285156,
"learning_rate": 5.555555555555555e-07,
"loss": 0.027,
"num_tokens": 454285.0,
"reward": -9.2822265625,
"reward_std": 5.530088424682617,
"rewards/rm_reward_func/mean": -9.2822265625,
"rewards/rm_reward_func/std": 6.309089183807373,
"step": 36
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 426.0,
"completions/mean_length": 222.09375,
"completions/mean_terminated_length": 212.74192810058594,
"completions/min_length": 56.0,
"completions/min_terminated_length": 56.0,
"epoch": 0.0296,
"grad_norm": 2.861426830291748,
"kl": 0.0002856254577636719,
"learning_rate": 5.714285714285714e-07,
"loss": -0.0668,
"num_tokens": 463344.0,
"reward": -7.4073486328125,
"reward_std": 6.660886287689209,
"rewards/rm_reward_func/mean": -7.4073486328125,
"rewards/rm_reward_func/std": 6.9068474769592285,
"step": 37
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 505.0,
"completions/mean_length": 188.90625,
"completions/mean_terminated_length": 98.43999481201172,
"completions/min_length": 12.0,
"completions/min_terminated_length": 12.0,
"epoch": 0.0304,
"grad_norm": 5.866414546966553,
"kl": 0.0003485679626464844,
"learning_rate": 5.873015873015873e-07,
"loss": -0.2332,
"num_tokens": 475037.0,
"reward": -9.38623046875,
"reward_std": 3.1565802097320557,
"rewards/rm_reward_func/mean": -9.38623046875,
"rewards/rm_reward_func/std": 10.67844295501709,
"step": 38
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 404.0,
"completions/mean_length": 303.28125,
"completions/mean_terminated_length": 221.60870361328125,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.0312,
"grad_norm": 2.542703151702881,
"kl": 0.00039768218994140625,
"learning_rate": 6.031746031746031e-07,
"loss": 0.076,
"num_tokens": 487278.0,
"reward": -8.93896484375,
"reward_std": 3.1694929599761963,
"rewards/rm_reward_func/mean": -8.93896484375,
"rewards/rm_reward_func/std": 8.222332000732422,
"step": 39
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 494.0,
"completions/mean_length": 229.0625,
"completions/mean_terminated_length": 163.7692413330078,
"completions/min_length": 9.0,
"completions/min_terminated_length": 9.0,
"epoch": 0.032,
"grad_norm": 6.272686004638672,
"kl": 0.0004992485046386719,
"learning_rate": 6.19047619047619e-07,
"loss": -0.0515,
"num_tokens": 498952.0,
"reward": -3.323486328125,
"reward_std": 5.879536151885986,
"rewards/rm_reward_func/mean": -3.323486328125,
"rewards/rm_reward_func/std": 8.639906883239746,
"step": 40
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 500.0,
"completions/mean_length": 375.125,
"completions/mean_terminated_length": 321.5652160644531,
"completions/min_length": 98.0,
"completions/min_terminated_length": 98.0,
"epoch": 0.0328,
"grad_norm": 2.0339434146881104,
"kl": 0.00039386749267578125,
"learning_rate": 6.349206349206349e-07,
"loss": -0.1041,
"num_tokens": 513076.0,
"reward": -4.98583984375,
"reward_std": 4.426126956939697,
"rewards/rm_reward_func/mean": -4.98583984375,
"rewards/rm_reward_func/std": 5.537177085876465,
"step": 41
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 397.0,
"completions/mean_length": 161.84375,
"completions/mean_terminated_length": 150.5483856201172,
"completions/min_length": 49.0,
"completions/min_terminated_length": 49.0,
"epoch": 0.0336,
"grad_norm": 3.7768754959106445,
"kl": 0.0003666877746582031,
"learning_rate": 6.507936507936507e-07,
"loss": -0.0783,
"num_tokens": 523351.0,
"reward": -8.8109130859375,
"reward_std": 5.987880706787109,
"rewards/rm_reward_func/mean": -8.8109130859375,
"rewards/rm_reward_func/std": 6.827143669128418,
"step": 42
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 206.0,
"completions/mean_length": 207.6875,
"completions/mean_terminated_length": 106.25,
"completions/min_length": 35.0,
"completions/min_terminated_length": 35.0,
"epoch": 0.0344,
"grad_norm": 4.852839469909668,
"kl": 0.0005655288696289062,
"learning_rate": 6.666666666666666e-07,
"loss": 0.0577,
"num_tokens": 532325.0,
"reward": -11.9395751953125,
"reward_std": 3.7294559478759766,
"rewards/rm_reward_func/mean": -11.9395751953125,
"rewards/rm_reward_func/std": 6.000690460205078,
"step": 43
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 493.0,
"completions/mean_length": 345.21875,
"completions/mean_terminated_length": 339.8387145996094,
"completions/min_length": 116.0,
"completions/min_terminated_length": 116.0,
"epoch": 0.0352,
"grad_norm": 2.18628191947937,
"kl": 0.0003714561462402344,
"learning_rate": 6.825396825396826e-07,
"loss": -0.0234,
"num_tokens": 546108.0,
"reward": -8.53497314453125,
"reward_std": 4.388243675231934,
"rewards/rm_reward_func/mean": -8.53497314453125,
"rewards/rm_reward_func/std": 7.794600486755371,
"step": 44
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 388.0,
"completions/mean_length": 305.90625,
"completions/mean_terminated_length": 197.95237731933594,
"completions/min_length": 53.0,
"completions/min_terminated_length": 53.0,
"epoch": 0.036,
"grad_norm": 2.3126137256622314,
"kl": 0.0004107952117919922,
"learning_rate": 6.984126984126983e-07,
"loss": 0.0486,
"num_tokens": 559873.0,
"reward": -7.420013427734375,
"reward_std": 7.450569152832031,
"rewards/rm_reward_func/mean": -7.420013427734375,
"rewards/rm_reward_func/std": 10.354388236999512,
"step": 45
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 387.25,
"completions/mean_terminated_length": 321.9047546386719,
"completions/min_length": 122.0,
"completions/min_terminated_length": 122.0,
"epoch": 0.0368,
"grad_norm": 2.344557762145996,
"kl": 0.00038242340087890625,
"learning_rate": 7.142857142857143e-07,
"loss": -0.011,
"num_tokens": 575289.0,
"reward": -4.837789535522461,
"reward_std": 7.047362327575684,
"rewards/rm_reward_func/mean": -4.837789535522461,
"rewards/rm_reward_func/std": 11.340645790100098,
"step": 46
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 233.0,
"completions/mean_length": 140.09375,
"completions/mean_terminated_length": 71.22222137451172,
"completions/min_length": 20.0,
"completions/min_terminated_length": 20.0,
"epoch": 0.0376,
"grad_norm": 5.920055389404297,
"kl": 0.0005631446838378906,
"learning_rate": 7.301587301587301e-07,
"loss": -0.1061,
"num_tokens": 586604.0,
"reward": -4.19140625,
"reward_std": 4.153356552124023,
"rewards/rm_reward_func/mean": -4.19140625,
"rewards/rm_reward_func/std": 7.545008182525635,
"step": 47
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 378.0,
"completions/max_terminated_length": 378.0,
"completions/mean_length": 163.6875,
"completions/mean_terminated_length": 163.6875,
"completions/min_length": 53.0,
"completions/min_terminated_length": 53.0,
"epoch": 0.0384,
"grad_norm": 3.746752977371216,
"kl": 0.0006389617919921875,
"learning_rate": 7.46031746031746e-07,
"loss": -0.094,
"num_tokens": 593642.0,
"reward": -10.421875,
"reward_std": 5.012988567352295,
"rewards/rm_reward_func/mean": -10.421875,
"rewards/rm_reward_func/std": 6.197994232177734,
"step": 48
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 500.0,
"completions/mean_length": 333.0625,
"completions/mean_terminated_length": 251.72727966308594,
"completions/min_length": 93.0,
"completions/min_terminated_length": 93.0,
"epoch": 0.0392,
"grad_norm": 3.094766139984131,
"kl": 0.0006709098815917969,
"learning_rate": 7.619047619047618e-07,
"loss": 0.0426,
"num_tokens": 606588.0,
"reward": -7.983154296875,
"reward_std": 8.31886100769043,
"rewards/rm_reward_func/mean": -7.983154296875,
"rewards/rm_reward_func/std": 10.532109260559082,
"step": 49
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 511.0,
"completions/mean_length": 229.5,
"completions/mean_terminated_length": 200.27586364746094,
"completions/min_length": 33.0,
"completions/min_terminated_length": 33.0,
"epoch": 0.04,
"grad_norm": 3.4759812355041504,
"kl": 0.0006818771362304688,
"learning_rate": 7.777777777777778e-07,
"loss": 0.1175,
"num_tokens": 616556.0,
"reward": -6.1270751953125,
"reward_std": 8.114751815795898,
"rewards/rm_reward_func/mean": -6.1270751953125,
"rewards/rm_reward_func/std": 8.840689659118652,
"step": 50
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 397.0,
"completions/mean_length": 311.0625,
"completions/mean_terminated_length": 205.8095245361328,
"completions/min_length": 47.0,
"completions/min_terminated_length": 47.0,
"epoch": 0.0408,
"grad_norm": 2.890885353088379,
"kl": 0.000766754150390625,
"learning_rate": 7.936507936507936e-07,
"loss": 0.0924,
"num_tokens": 628694.0,
"reward": -8.6982421875,
"reward_std": 3.657209873199463,
"rewards/rm_reward_func/mean": -8.6982421875,
"rewards/rm_reward_func/std": 13.067253112792969,
"step": 51
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 489.0,
"completions/mean_length": 283.28125,
"completions/mean_terminated_length": 163.4761962890625,
"completions/min_length": 27.0,
"completions/min_terminated_length": 27.0,
"epoch": 0.0416,
"grad_norm": 2.9035823345184326,
"kl": 0.0004100799560546875,
"learning_rate": 8.095238095238095e-07,
"loss": 0.2123,
"num_tokens": 641223.0,
"reward": -5.83807373046875,
"reward_std": 4.900543689727783,
"rewards/rm_reward_func/mean": -5.83807373046875,
"rewards/rm_reward_func/std": 7.653470993041992,
"step": 52
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 425.0,
"completions/max_terminated_length": 425.0,
"completions/mean_length": 114.9375,
"completions/mean_terminated_length": 114.9375,
"completions/min_length": 6.0,
"completions/min_terminated_length": 6.0,
"epoch": 0.0424,
"grad_norm": 10.671420097351074,
"kl": 0.0011627674102783203,
"learning_rate": 8.253968253968253e-07,
"loss": -0.1026,
"num_tokens": 647501.0,
"reward": -8.42413330078125,
"reward_std": 5.442896842956543,
"rewards/rm_reward_func/mean": -8.42413330078125,
"rewards/rm_reward_func/std": 8.57506275177002,
"step": 53
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 468.0,
"completions/mean_length": 252.53125,
"completions/mean_terminated_length": 179.87998962402344,
"completions/min_length": 41.0,
"completions/min_terminated_length": 41.0,
"epoch": 0.0432,
"grad_norm": 3.140366315841675,
"kl": 0.0007600784301757812,
"learning_rate": 8.412698412698413e-07,
"loss": 0.1312,
"num_tokens": 660070.0,
"reward": -10.078125,
"reward_std": 4.523303985595703,
"rewards/rm_reward_func/mean": -10.078125,
"rewards/rm_reward_func/std": 6.438615322113037,
"step": 54
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 490.0,
"completions/mean_length": 202.15625,
"completions/mean_terminated_length": 157.8928680419922,
"completions/min_length": 21.0,
"completions/min_terminated_length": 21.0,
"epoch": 0.044,
"grad_norm": 4.555398941040039,
"kl": 0.0013036727905273438,
"learning_rate": 8.57142857142857e-07,
"loss": 0.0684,
"num_tokens": 669715.0,
"reward": -10.841552734375,
"reward_std": 8.45435619354248,
"rewards/rm_reward_func/mean": -10.841552734375,
"rewards/rm_reward_func/std": 9.872593879699707,
"step": 55
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.4375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 479.0,
"completions/mean_length": 336.65625,
"completions/mean_terminated_length": 200.2777862548828,
"completions/min_length": 30.0,
"completions/min_terminated_length": 30.0,
"epoch": 0.0448,
"grad_norm": 2.7749080657958984,
"kl": 0.0010623931884765625,
"learning_rate": 8.73015873015873e-07,
"loss": -0.0338,
"num_tokens": 683152.0,
"reward": -1.090301513671875,
"reward_std": 3.921980142593384,
"rewards/rm_reward_func/mean": -1.090301513671875,
"rewards/rm_reward_func/std": 6.629665374755859,
"step": 56
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 501.0,
"completions/mean_length": 320.03125,
"completions/mean_terminated_length": 244.91305541992188,
"completions/min_length": 20.0,
"completions/min_terminated_length": 20.0,
"epoch": 0.0456,
"grad_norm": 2.539121389389038,
"kl": 0.0009260177612304688,
"learning_rate": 8.888888888888888e-07,
"loss": 0.1425,
"num_tokens": 697105.0,
"reward": -7.415740966796875,
"reward_std": 5.969917297363281,
"rewards/rm_reward_func/mean": -7.415740966796875,
"rewards/rm_reward_func/std": 7.336832523345947,
"step": 57
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.40625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 505.0,
"completions/mean_length": 330.71875,
"completions/mean_terminated_length": 206.68421936035156,
"completions/min_length": 49.0,
"completions/min_terminated_length": 49.0,
"epoch": 0.0464,
"grad_norm": 2.4101502895355225,
"kl": 0.0011167526245117188,
"learning_rate": 9.047619047619047e-07,
"loss": -0.0899,
"num_tokens": 710128.0,
"reward": -6.55548095703125,
"reward_std": 5.115406036376953,
"rewards/rm_reward_func/mean": -6.55548095703125,
"rewards/rm_reward_func/std": 6.5659332275390625,
"step": 58
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 307.375,
"completions/mean_terminated_length": 239.1666717529297,
"completions/min_length": 39.0,
"completions/min_terminated_length": 39.0,
"epoch": 0.0472,
"grad_norm": 2.872269630432129,
"kl": 0.0010232925415039062,
"learning_rate": 9.206349206349205e-07,
"loss": 0.014,
"num_tokens": 721980.0,
"reward": -5.94189453125,
"reward_std": 5.278933525085449,
"rewards/rm_reward_func/mean": -5.94189453125,
"rewards/rm_reward_func/std": 8.825187683105469,
"step": 59
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 475.0,
"completions/mean_length": 270.4375,
"completions/mean_terminated_length": 235.9285888671875,
"completions/min_length": 49.0,
"completions/min_terminated_length": 49.0,
"epoch": 0.048,
"grad_norm": 3.2065460681915283,
"kl": 0.0014667510986328125,
"learning_rate": 9.365079365079365e-07,
"loss": 0.2253,
"num_tokens": 733306.0,
"reward": -5.34954833984375,
"reward_std": 8.191361427307129,
"rewards/rm_reward_func/mean": -5.34954833984375,
"rewards/rm_reward_func/std": 10.043052673339844,
"step": 60
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 470.0,
"completions/mean_length": 302.96875,
"completions/mean_terminated_length": 254.73077392578125,
"completions/min_length": 10.0,
"completions/min_terminated_length": 10.0,
"epoch": 0.0488,
"grad_norm": 2.4722800254821777,
"kl": 0.0020198822021484375,
"learning_rate": 9.523809523809522e-07,
"loss": -0.0115,
"num_tokens": 745889.0,
"reward": -13.30328369140625,
"reward_std": 3.812379837036133,
"rewards/rm_reward_func/mean": -13.30328369140625,
"rewards/rm_reward_func/std": 8.161885261535645,
"step": 61
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 501.0,
"completions/mean_length": 218.5625,
"completions/mean_terminated_length": 209.09677124023438,
"completions/min_length": 17.0,
"completions/min_terminated_length": 17.0,
"epoch": 0.0496,
"grad_norm": 3.010167360305786,
"kl": 0.0018796920776367188,
"learning_rate": 9.682539682539682e-07,
"loss": -0.2943,
"num_tokens": 754907.0,
"reward": -7.835418701171875,
"reward_std": 5.403800010681152,
"rewards/rm_reward_func/mean": -7.835418701171875,
"rewards/rm_reward_func/std": 8.523995399475098,
"step": 62
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 455.0,
"completions/mean_length": 331.15625,
"completions/mean_terminated_length": 248.95455932617188,
"completions/min_length": 92.0,
"completions/min_terminated_length": 92.0,
"epoch": 0.0504,
"grad_norm": 2.819601058959961,
"kl": 0.0015172958374023438,
"learning_rate": 9.84126984126984e-07,
"loss": -0.0081,
"num_tokens": 768560.0,
"reward": -11.4364013671875,
"reward_std": 3.361478328704834,
"rewards/rm_reward_func/mean": -11.4364013671875,
"rewards/rm_reward_func/std": 6.750278472900391,
"step": 63
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 485.0,
"completions/mean_length": 327.03125,
"completions/mean_terminated_length": 230.1428680419922,
"completions/min_length": 16.0,
"completions/min_terminated_length": 16.0,
"epoch": 0.0512,
"grad_norm": 3.7442524433135986,
"kl": 0.0022602081298828125,
"learning_rate": 1e-06,
"loss": -0.1989,
"num_tokens": 781345.0,
"reward": -8.39996337890625,
"reward_std": 4.888169288635254,
"rewards/rm_reward_func/mean": -8.39996337890625,
"rewards/rm_reward_func/std": 5.147318363189697,
"step": 64
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 268.40625,
"completions/mean_terminated_length": 252.16668701171875,
"completions/min_length": 73.0,
"completions/min_terminated_length": 73.0,
"epoch": 0.052,
"grad_norm": 2.8628954887390137,
"kl": 0.0019245147705078125,
"learning_rate": 1e-06,
"loss": -0.0962,
"num_tokens": 792750.0,
"reward": -7.360107421875,
"reward_std": 5.436917781829834,
"rewards/rm_reward_func/mean": -7.360107421875,
"rewards/rm_reward_func/std": 11.22769832611084,
"step": 65
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.40625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 493.0,
"completions/mean_length": 332.46875,
"completions/mean_terminated_length": 209.63157653808594,
"completions/min_length": 25.0,
"completions/min_terminated_length": 25.0,
"epoch": 0.0528,
"grad_norm": 3.654360294342041,
"kl": 0.0015087127685546875,
"learning_rate": 1e-06,
"loss": -0.1237,
"num_tokens": 805853.0,
"reward": -9.6962890625,
"reward_std": 2.946159839630127,
"rewards/rm_reward_func/mean": -9.6962890625,
"rewards/rm_reward_func/std": 4.295775890350342,
"step": 66
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.53125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 502.0,
"completions/mean_length": 402.0,
"completions/mean_terminated_length": 277.3333435058594,
"completions/min_length": 82.0,
"completions/min_terminated_length": 82.0,
"epoch": 0.0536,
"grad_norm": 2.3473589420318604,
"kl": 0.002231597900390625,
"learning_rate": 1e-06,
"loss": -0.0437,
"num_tokens": 823173.0,
"reward": -8.37060546875,
"reward_std": 3.5887060165405273,
"rewards/rm_reward_func/mean": -8.37060546875,
"rewards/rm_reward_func/std": 4.726972579956055,
"step": 67
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.4375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 488.0,
"completions/mean_length": 364.6875,
"completions/mean_terminated_length": 250.11111450195312,
"completions/min_length": 38.0,
"completions/min_terminated_length": 38.0,
"epoch": 0.0544,
"grad_norm": 2.694181442260742,
"kl": 0.001918792724609375,
"learning_rate": 1e-06,
"loss": 0.2082,
"num_tokens": 837739.0,
"reward": -7.8717041015625,
"reward_std": 3.8687283992767334,
"rewards/rm_reward_func/mean": -7.8717041015625,
"rewards/rm_reward_func/std": 7.253809928894043,
"step": 68
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 490.0,
"completions/mean_length": 312.46875,
"completions/mean_terminated_length": 256.6000061035156,
"completions/min_length": 85.0,
"completions/min_terminated_length": 85.0,
"epoch": 0.0552,
"grad_norm": 2.733241558074951,
"kl": 0.002780914306640625,
"learning_rate": 1e-06,
"loss": -0.0181,
"num_tokens": 849434.0,
"reward": -6.937103271484375,
"reward_std": 3.667171001434326,
"rewards/rm_reward_func/mean": -6.937103271484375,
"rewards/rm_reward_func/std": 5.047467231750488,
"step": 69
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 507.0,
"completions/mean_length": 336.71875,
"completions/mean_terminated_length": 278.29168701171875,
"completions/min_length": 54.0,
"completions/min_terminated_length": 54.0,
"epoch": 0.056,
"grad_norm": 2.4156734943389893,
"kl": 0.002368927001953125,
"learning_rate": 1e-06,
"loss": -0.1443,
"num_tokens": 862321.0,
"reward": -4.07220458984375,
"reward_std": 10.273677825927734,
"rewards/rm_reward_func/mean": -4.07220458984375,
"rewards/rm_reward_func/std": 13.013540267944336,
"step": 70
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.40625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 419.0,
"completions/mean_length": 314.90625,
"completions/mean_terminated_length": 180.05262756347656,
"completions/min_length": 70.0,
"completions/min_terminated_length": 70.0,
"epoch": 0.0568,
"grad_norm": 3.1217379570007324,
"kl": 0.00289154052734375,
"learning_rate": 1e-06,
"loss": -0.0551,
"num_tokens": 875374.0,
"reward": -6.78515625,
"reward_std": 4.401189804077148,
"rewards/rm_reward_func/mean": -6.78515625,
"rewards/rm_reward_func/std": 7.016132354736328,
"step": 71
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 502.0,
"completions/mean_length": 353.125,
"completions/mean_terminated_length": 323.7037048339844,
"completions/min_length": 55.0,
"completions/min_terminated_length": 55.0,
"epoch": 0.0576,
"grad_norm": 2.9577646255493164,
"kl": 0.003635406494140625,
"learning_rate": 1e-06,
"loss": -0.1052,
"num_tokens": 889138.0,
"reward": -3.15277099609375,
"reward_std": 5.309854507446289,
"rewards/rm_reward_func/mean": -3.15277099609375,
"rewards/rm_reward_func/std": 5.763915538787842,
"step": 72
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 466.0,
"completions/mean_length": 258.90625,
"completions/mean_terminated_length": 188.0399932861328,
"completions/min_length": 18.0,
"completions/min_terminated_length": 18.0,
"epoch": 0.0584,
"grad_norm": 3.2543458938598633,
"kl": 0.0028629302978515625,
"learning_rate": 1e-06,
"loss": 0.2214,
"num_tokens": 900951.0,
"reward": -8.47222900390625,
"reward_std": 4.21871280670166,
"rewards/rm_reward_func/mean": -8.47222900390625,
"rewards/rm_reward_func/std": 5.818933486938477,
"step": 73
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 414.0,
"completions/mean_length": 243.0,
"completions/mean_terminated_length": 153.33334350585938,
"completions/min_length": 32.0,
"completions/min_terminated_length": 32.0,
"epoch": 0.0592,
"grad_norm": 3.68485426902771,
"kl": 0.0032253265380859375,
"learning_rate": 1e-06,
"loss": -0.0463,
"num_tokens": 911351.0,
"reward": -9.206695556640625,
"reward_std": 6.0724778175354,
"rewards/rm_reward_func/mean": -9.206695556640625,
"rewards/rm_reward_func/std": 7.073054313659668,
"step": 74
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 472.0,
"completions/mean_length": 289.78125,
"completions/mean_terminated_length": 238.50001525878906,
"completions/min_length": 44.0,
"completions/min_terminated_length": 44.0,
"epoch": 0.06,
"grad_norm": 2.78497052192688,
"kl": 0.00466156005859375,
"learning_rate": 1e-06,
"loss": 0.2228,
"num_tokens": 923680.0,
"reward": -6.83343505859375,
"reward_std": 7.71270227432251,
"rewards/rm_reward_func/mean": -6.83343505859375,
"rewards/rm_reward_func/std": 10.820587158203125,
"step": 75
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.71875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 497.0,
"completions/mean_length": 481.46875,
"completions/mean_terminated_length": 403.4444580078125,
"completions/min_length": 134.0,
"completions/min_terminated_length": 134.0,
"epoch": 0.0608,
"grad_norm": 1.8886445760726929,
"kl": 0.002117633819580078,
"learning_rate": 1e-06,
"loss": 0.0654,
"num_tokens": 944191.0,
"reward": -10.4248046875,
"reward_std": 5.140851020812988,
"rewards/rm_reward_func/mean": -10.4248046875,
"rewards/rm_reward_func/std": 9.057696342468262,
"step": 76
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 459.0,
"completions/mean_length": 180.1875,
"completions/mean_terminated_length": 169.48387145996094,
"completions/min_length": 9.0,
"completions/min_terminated_length": 9.0,
"epoch": 0.0616,
"grad_norm": 5.139257431030273,
"kl": 0.002521514892578125,
"learning_rate": 1e-06,
"loss": 0.1781,
"num_tokens": 952477.0,
"reward": -1.8204498291015625,
"reward_std": 4.737975120544434,
"rewards/rm_reward_func/mean": -1.8204498291015625,
"rewards/rm_reward_func/std": 12.699488639831543,
"step": 77
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 411.0,
"completions/mean_length": 212.5625,
"completions/mean_terminated_length": 181.58621215820312,
"completions/min_length": 33.0,
"completions/min_terminated_length": 33.0,
"epoch": 0.0624,
"grad_norm": 3.0258443355560303,
"kl": 0.0031642913818359375,
"learning_rate": 1e-06,
"loss": 0.2051,
"num_tokens": 963415.0,
"reward": -5.3250732421875,
"reward_std": 4.80600643157959,
"rewards/rm_reward_func/mean": -5.3250732421875,
"rewards/rm_reward_func/std": 4.75626802444458,
"step": 78
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 497.0,
"completions/mean_length": 343.53125,
"completions/mean_terminated_length": 277.60870361328125,
"completions/min_length": 40.0,
"completions/min_terminated_length": 40.0,
"epoch": 0.0632,
"grad_norm": 2.744492292404175,
"kl": 0.00372314453125,
"learning_rate": 1e-06,
"loss": -0.0128,
"num_tokens": 977392.0,
"reward": -4.4423980712890625,
"reward_std": 5.815633773803711,
"rewards/rm_reward_func/mean": -4.4423980712890625,
"rewards/rm_reward_func/std": 10.560279846191406,
"step": 79
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.4375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 466.0,
"completions/mean_length": 312.25,
"completions/mean_terminated_length": 156.88888549804688,
"completions/min_length": 51.0,
"completions/min_terminated_length": 51.0,
"epoch": 0.064,
"grad_norm": 2.8983473777770996,
"kl": 0.003662109375,
"learning_rate": 1e-06,
"loss": 0.042,
"num_tokens": 995120.0,
"reward": -12.873046875,
"reward_std": 3.767939805984497,
"rewards/rm_reward_func/mean": -12.873046875,
"rewards/rm_reward_func/std": 7.584336757659912,
"step": 80
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 413.0,
"completions/mean_length": 246.96875,
"completions/mean_terminated_length": 185.8076934814453,
"completions/min_length": 31.0,
"completions/min_terminated_length": 31.0,
"epoch": 0.0648,
"grad_norm": 3.573094129562378,
"kl": 0.0052947998046875,
"learning_rate": 1e-06,
"loss": -0.1861,
"num_tokens": 1005943.0,
"reward": -5.154052734375,
"reward_std": 5.333109378814697,
"rewards/rm_reward_func/mean": -5.154052734375,
"rewards/rm_reward_func/std": 5.350836277008057,
"step": 81
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 470.0,
"completions/mean_length": 298.96875,
"completions/mean_terminated_length": 249.80770874023438,
"completions/min_length": 38.0,
"completions/min_terminated_length": 38.0,
"epoch": 0.0656,
"grad_norm": 2.644007444381714,
"kl": 0.003917694091796875,
"learning_rate": 1e-06,
"loss": 0.0187,
"num_tokens": 1020238.0,
"reward": -9.076370239257812,
"reward_std": 4.40856409072876,
"rewards/rm_reward_func/mean": -9.076370239257812,
"rewards/rm_reward_func/std": 6.288516521453857,
"step": 82
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 465.0,
"completions/mean_length": 307.0,
"completions/mean_terminated_length": 199.61904907226562,
"completions/min_length": 56.0,
"completions/min_terminated_length": 56.0,
"epoch": 0.0664,
"grad_norm": 2.8343067169189453,
"kl": 0.0068359375,
"learning_rate": 1e-06,
"loss": -0.0099,
"num_tokens": 1032262.0,
"reward": -7.667877197265625,
"reward_std": 5.629947185516357,
"rewards/rm_reward_func/mean": -7.667877197265625,
"rewards/rm_reward_func/std": 7.898858070373535,
"step": 83
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 465.0,
"completions/max_terminated_length": 465.0,
"completions/mean_length": 225.71875,
"completions/mean_terminated_length": 225.71875,
"completions/min_length": 93.0,
"completions/min_terminated_length": 93.0,
"epoch": 0.0672,
"grad_norm": 2.8555967807769775,
"kl": 0.00545501708984375,
"learning_rate": 1e-06,
"loss": 0.0421,
"num_tokens": 1042485.0,
"reward": 2.5863037109375,
"reward_std": 3.6911566257476807,
"rewards/rm_reward_func/mean": 2.5863037109375,
"rewards/rm_reward_func/std": 10.012650489807129,
"step": 84
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 435.0,
"completions/mean_length": 219.59375,
"completions/mean_terminated_length": 210.16128540039062,
"completions/min_length": 25.0,
"completions/min_terminated_length": 25.0,
"epoch": 0.068,
"grad_norm": 4.213369369506836,
"kl": 0.00635528564453125,
"learning_rate": 1e-06,
"loss": 0.0863,
"num_tokens": 1052608.0,
"reward": -3.3102264404296875,
"reward_std": 3.797806978225708,
"rewards/rm_reward_func/mean": -3.3102264404296875,
"rewards/rm_reward_func/std": 5.886042594909668,
"step": 85
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 289.65625,
"completions/mean_terminated_length": 266.6551818847656,
"completions/min_length": 59.0,
"completions/min_terminated_length": 59.0,
"epoch": 0.0688,
"grad_norm": 3.4353296756744385,
"kl": 0.0052337646484375,
"learning_rate": 1e-06,
"loss": 0.0776,
"num_tokens": 1064949.0,
"reward": 2.492401123046875,
"reward_std": 7.244277000427246,
"rewards/rm_reward_func/mean": 2.492401123046875,
"rewards/rm_reward_func/std": 9.47168254852295,
"step": 86
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 498.0,
"completions/mean_length": 236.28125,
"completions/mean_terminated_length": 196.8928680419922,
"completions/min_length": 9.0,
"completions/min_terminated_length": 9.0,
"epoch": 0.0696,
"grad_norm": 9.39013385772705,
"kl": 0.00577545166015625,
"learning_rate": 1e-06,
"loss": -0.0612,
"num_tokens": 1074950.0,
"reward": -5.302734375,
"reward_std": 6.743742942810059,
"rewards/rm_reward_func/mean": -5.302734375,
"rewards/rm_reward_func/std": 8.008939743041992,
"step": 87
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 474.0,
"completions/mean_length": 292.46875,
"completions/mean_terminated_length": 192.68182373046875,
"completions/min_length": 26.0,
"completions/min_terminated_length": 26.0,
"epoch": 0.0704,
"grad_norm": 3.4955945014953613,
"kl": 0.0057942867279052734,
"learning_rate": 1e-06,
"loss": 0.2027,
"num_tokens": 1090957.0,
"reward": -4.89019775390625,
"reward_std": 5.86878776550293,
"rewards/rm_reward_func/mean": -4.89019775390625,
"rewards/rm_reward_func/std": 11.281463623046875,
"step": 88
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.40625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 501.0,
"completions/mean_length": 299.4375,
"completions/mean_terminated_length": 154.0,
"completions/min_length": 36.0,
"completions/min_terminated_length": 36.0,
"epoch": 0.0712,
"grad_norm": 3.3145103454589844,
"kl": 0.00629425048828125,
"learning_rate": 1e-06,
"loss": 0.0759,
"num_tokens": 1103699.0,
"reward": -7.9558868408203125,
"reward_std": 3.8132681846618652,
"rewards/rm_reward_func/mean": -7.9558868408203125,
"rewards/rm_reward_func/std": 5.60645055770874,
"step": 89
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 462.0,
"completions/mean_length": 242.3125,
"completions/mean_terminated_length": 214.41378784179688,
"completions/min_length": 46.0,
"completions/min_terminated_length": 46.0,
"epoch": 0.072,
"grad_norm": 3.746366500854492,
"kl": 0.007389068603515625,
"learning_rate": 1e-06,
"loss": 0.0806,
"num_tokens": 1114293.0,
"reward": 4.7412109375,
"reward_std": 6.3503098487854,
"rewards/rm_reward_func/mean": 4.7412109375,
"rewards/rm_reward_func/std": 12.657453536987305,
"step": 90
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.59375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 463.0,
"completions/mean_length": 392.34375,
"completions/mean_terminated_length": 217.4615478515625,
"completions/min_length": 42.0,
"completions/min_terminated_length": 42.0,
"epoch": 0.0728,
"grad_norm": 2.246333360671997,
"kl": 0.00717926025390625,
"learning_rate": 1e-06,
"loss": -0.0118,
"num_tokens": 1130400.0,
"reward": -6.076568603515625,
"reward_std": 4.733026027679443,
"rewards/rm_reward_func/mean": -6.076568603515625,
"rewards/rm_reward_func/std": 10.103605270385742,
"step": 91
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 402.0,
"completions/mean_length": 368.15625,
"completions/mean_terminated_length": 292.8095397949219,
"completions/min_length": 19.0,
"completions/min_terminated_length": 19.0,
"epoch": 0.0736,
"grad_norm": 2.1302545070648193,
"kl": 0.007269859313964844,
"learning_rate": 1e-06,
"loss": 0.1118,
"num_tokens": 1146957.0,
"reward": -7.0111083984375,
"reward_std": 6.281430244445801,
"rewards/rm_reward_func/mean": -7.0111083984375,
"rewards/rm_reward_func/std": 10.263130187988281,
"step": 92
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 502.0,
"completions/mean_length": 339.0,
"completions/mean_terminated_length": 271.3043518066406,
"completions/min_length": 32.0,
"completions/min_terminated_length": 32.0,
"epoch": 0.0744,
"grad_norm": 3.688321113586426,
"kl": 0.00998687744140625,
"learning_rate": 1e-06,
"loss": 0.0504,
"num_tokens": 1159773.0,
"reward": -3.8799972534179688,
"reward_std": 4.736905097961426,
"rewards/rm_reward_func/mean": -3.8799972534179688,
"rewards/rm_reward_func/std": 6.819652557373047,
"step": 93
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 484.0,
"completions/mean_length": 260.21875,
"completions/mean_terminated_length": 252.09677124023438,
"completions/min_length": 64.0,
"completions/min_terminated_length": 64.0,
"epoch": 0.0752,
"grad_norm": 3.679842472076416,
"kl": 0.00600433349609375,
"learning_rate": 1e-06,
"loss": -0.1287,
"num_tokens": 1171236.0,
"reward": -5.69622802734375,
"reward_std": 3.642548084259033,
"rewards/rm_reward_func/mean": -5.69622802734375,
"rewards/rm_reward_func/std": 5.160977363586426,
"step": 94
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.40625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 376.25,
"completions/mean_terminated_length": 283.368408203125,
"completions/min_length": 32.0,
"completions/min_terminated_length": 32.0,
"epoch": 0.076,
"grad_norm": 2.582540273666382,
"kl": 0.00710296630859375,
"learning_rate": 1e-06,
"loss": -0.1972,
"num_tokens": 1186820.0,
"reward": -8.5262451171875,
"reward_std": 3.4011950492858887,
"rewards/rm_reward_func/mean": -8.5262451171875,
"rewards/rm_reward_func/std": 7.117488384246826,
"step": 95
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 423.0,
"completions/mean_length": 178.6875,
"completions/mean_terminated_length": 167.93548583984375,
"completions/min_length": 23.0,
"completions/min_terminated_length": 23.0,
"epoch": 0.0768,
"grad_norm": 4.087407112121582,
"kl": 0.0051116943359375,
"learning_rate": 1e-06,
"loss": -0.1202,
"num_tokens": 1198074.0,
"reward": -5.597259521484375,
"reward_std": 7.2116804122924805,
"rewards/rm_reward_func/mean": -5.597259521484375,
"rewards/rm_reward_func/std": 8.06857681274414,
"step": 96
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 488.0,
"completions/mean_length": 284.03125,
"completions/mean_terminated_length": 208.0416717529297,
"completions/min_length": 39.0,
"completions/min_terminated_length": 39.0,
"epoch": 0.0776,
"grad_norm": 9.104924201965332,
"kl": 0.00815582275390625,
"learning_rate": 1e-06,
"loss": -0.1018,
"num_tokens": 1211131.0,
"reward": -9.880126953125,
"reward_std": 6.673112869262695,
"rewards/rm_reward_func/mean": -9.880126953125,
"rewards/rm_reward_func/std": 7.890270233154297,
"step": 97
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 318.0,
"completions/mean_length": 260.78125,
"completions/mean_terminated_length": 177.0416717529297,
"completions/min_length": 53.0,
"completions/min_terminated_length": 53.0,
"epoch": 0.0784,
"grad_norm": 5.061870098114014,
"kl": 0.0068988800048828125,
"learning_rate": 1e-06,
"loss": 0.1266,
"num_tokens": 1224396.0,
"reward": -9.5322265625,
"reward_std": 5.915492057800293,
"rewards/rm_reward_func/mean": -9.5322265625,
"rewards/rm_reward_func/std": 13.454625129699707,
"step": 98
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 312.5,
"completions/mean_terminated_length": 275.5555725097656,
"completions/min_length": 15.0,
"completions/min_terminated_length": 15.0,
"epoch": 0.0792,
"grad_norm": 7.630608081817627,
"kl": 0.008739471435546875,
"learning_rate": 1e-06,
"loss": -0.0647,
"num_tokens": 1239732.0,
"reward": 1.023895263671875,
"reward_std": 8.057183265686035,
"rewards/rm_reward_func/mean": 1.023895263671875,
"rewards/rm_reward_func/std": 10.845428466796875,
"step": 99
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 415.0,
"completions/max_terminated_length": 415.0,
"completions/mean_length": 252.8125,
"completions/mean_terminated_length": 252.8125,
"completions/min_length": 49.0,
"completions/min_terminated_length": 49.0,
"epoch": 0.08,
"grad_norm": 3.1217799186706543,
"kl": 0.01123046875,
"learning_rate": 1e-06,
"loss": -0.0374,
"num_tokens": 1250262.0,
"reward": -8.40460205078125,
"reward_std": 3.330627679824829,
"rewards/rm_reward_func/mean": -8.40460205078125,
"rewards/rm_reward_func/std": 4.349534034729004,
"step": 100
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 329.0,
"completions/mean_length": 245.09375,
"completions/mean_terminated_length": 123.7727279663086,
"completions/min_length": 11.0,
"completions/min_terminated_length": 11.0,
"epoch": 0.0808,
"grad_norm": 4.933348655700684,
"kl": 0.0134124755859375,
"learning_rate": 1e-06,
"loss": 0.1239,
"num_tokens": 1260473.0,
"reward": -4.2023468017578125,
"reward_std": 4.068781852722168,
"rewards/rm_reward_func/mean": -4.2023468017578125,
"rewards/rm_reward_func/std": 4.7899885177612305,
"step": 101
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 494.0,
"completions/mean_length": 316.03125,
"completions/mean_terminated_length": 270.8077087402344,
"completions/min_length": 39.0,
"completions/min_terminated_length": 39.0,
"epoch": 0.0816,
"grad_norm": 5.183587551116943,
"kl": 0.01630401611328125,
"learning_rate": 1e-06,
"loss": -0.0734,
"num_tokens": 1272762.0,
"reward": -2.0743408203125,
"reward_std": 4.2065324783325195,
"rewards/rm_reward_func/mean": -2.0743408203125,
"rewards/rm_reward_func/std": 7.418477535247803,
"step": 102
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 502.0,
"completions/mean_length": 190.34375,
"completions/mean_terminated_length": 144.3928680419922,
"completions/min_length": 36.0,
"completions/min_terminated_length": 36.0,
"epoch": 0.0824,
"grad_norm": 9.615796089172363,
"kl": 0.0133819580078125,
"learning_rate": 1e-06,
"loss": 0.2496,
"num_tokens": 1282085.0,
"reward": -2.1240234375,
"reward_std": 6.015683650970459,
"rewards/rm_reward_func/mean": -2.1240234375,
"rewards/rm_reward_func/std": 7.95778751373291,
"step": 103
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.4375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 505.0,
"completions/mean_length": 318.75,
"completions/mean_terminated_length": 168.44444274902344,
"completions/min_length": 21.0,
"completions/min_terminated_length": 21.0,
"epoch": 0.0832,
"grad_norm": 4.339044094085693,
"kl": 0.01284027099609375,
"learning_rate": 1e-06,
"loss": 0.0598,
"num_tokens": 1298069.0,
"reward": -1.1455078125,
"reward_std": 5.1649861335754395,
"rewards/rm_reward_func/mean": -1.1455078125,
"rewards/rm_reward_func/std": 7.434368133544922,
"step": 104
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.65625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 491.0,
"completions/mean_length": 463.78125,
"completions/mean_terminated_length": 371.727294921875,
"completions/min_length": 202.0,
"completions/min_terminated_length": 202.0,
"epoch": 0.084,
"grad_norm": 3.03131365776062,
"kl": 0.0128326416015625,
"learning_rate": 1e-06,
"loss": -0.0067,
"num_tokens": 1315710.0,
"reward": -4.12103271484375,
"reward_std": 5.871215343475342,
"rewards/rm_reward_func/mean": -4.12103271484375,
"rewards/rm_reward_func/std": 8.927309036254883,
"step": 105
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.40625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 449.0,
"completions/mean_length": 384.53125,
"completions/mean_terminated_length": 297.3157958984375,
"completions/min_length": 50.0,
"completions/min_terminated_length": 50.0,
"epoch": 0.0848,
"grad_norm": 2.725952625274658,
"kl": 0.010955810546875,
"learning_rate": 1e-06,
"loss": 0.0927,
"num_tokens": 1331855.0,
"reward": -1.3194122314453125,
"reward_std": 3.2353854179382324,
"rewards/rm_reward_func/mean": -1.3194122314453125,
"rewards/rm_reward_func/std": 4.594810485839844,
"step": 106
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 463.0,
"completions/mean_length": 315.5625,
"completions/mean_terminated_length": 250.08334350585938,
"completions/min_length": 54.0,
"completions/min_terminated_length": 54.0,
"epoch": 0.0856,
"grad_norm": 29.571395874023438,
"kl": 0.010345458984375,
"learning_rate": 1e-06,
"loss": -0.1343,
"num_tokens": 1345001.0,
"reward": -6.830474853515625,
"reward_std": 4.087754726409912,
"rewards/rm_reward_func/mean": -6.830474853515625,
"rewards/rm_reward_func/std": 9.476208686828613,
"step": 107
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 505.0,
"completions/mean_length": 291.5625,
"completions/mean_terminated_length": 240.69232177734375,
"completions/min_length": 27.0,
"completions/min_terminated_length": 27.0,
"epoch": 0.0864,
"grad_norm": 4.20673131942749,
"kl": 0.013458251953125,
"learning_rate": 1e-06,
"loss": 0.1504,
"num_tokens": 1357331.0,
"reward": -5.711700439453125,
"reward_std": 3.702240467071533,
"rewards/rm_reward_func/mean": -5.711700439453125,
"rewards/rm_reward_func/std": 7.0592451095581055,
"step": 108
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 483.0,
"completions/mean_length": 364.8125,
"completions/mean_terminated_length": 297.9090881347656,
"completions/min_length": 112.0,
"completions/min_terminated_length": 112.0,
"epoch": 0.0872,
"grad_norm": 2.5211193561553955,
"kl": 0.0170440673828125,
"learning_rate": 1e-06,
"loss": -0.0797,
"num_tokens": 1371525.0,
"reward": 3.977081298828125,
"reward_std": 5.856342792510986,
"rewards/rm_reward_func/mean": 3.977081298828125,
"rewards/rm_reward_func/std": 12.513916969299316,
"step": 109
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 464.0,
"completions/mean_length": 319.65625,
"completions/mean_terminated_length": 255.5416717529297,
"completions/min_length": 122.0,
"completions/min_terminated_length": 122.0,
"epoch": 0.088,
"grad_norm": 4.921560287475586,
"kl": 0.01397705078125,
"learning_rate": 1e-06,
"loss": 0.0172,
"num_tokens": 1384498.0,
"reward": -5.39752197265625,
"reward_std": 3.883021354675293,
"rewards/rm_reward_func/mean": -5.39752197265625,
"rewards/rm_reward_func/std": 11.68447208404541,
"step": 110
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 317.4375,
"completions/mean_terminated_length": 215.52381896972656,
"completions/min_length": 27.0,
"completions/min_terminated_length": 27.0,
"epoch": 0.0888,
"grad_norm": 5.259191989898682,
"kl": 0.0244293212890625,
"learning_rate": 1e-06,
"loss": 0.0037,
"num_tokens": 1398792.0,
"reward": -9.126190185546875,
"reward_std": 3.321913719177246,
"rewards/rm_reward_func/mean": -9.126190185546875,
"rewards/rm_reward_func/std": 8.484477043151855,
"step": 111
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.5,
"completions/max_length": 512.0,
"completions/max_terminated_length": 411.0,
"completions/mean_length": 360.4375,
"completions/mean_terminated_length": 208.875,
"completions/min_length": 35.0,
"completions/min_terminated_length": 35.0,
"epoch": 0.0896,
"grad_norm": 13.804244995117188,
"kl": 0.0152130126953125,
"learning_rate": 1e-06,
"loss": 0.132,
"num_tokens": 1413758.0,
"reward": -12.7744140625,
"reward_std": 3.5870838165283203,
"rewards/rm_reward_func/mean": -12.7744140625,
"rewards/rm_reward_func/std": 4.53514289855957,
"step": 112
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.4375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 399.8125,
"completions/mean_terminated_length": 312.5555725097656,
"completions/min_length": 150.0,
"completions/min_terminated_length": 150.0,
"epoch": 0.0904,
"grad_norm": 2.4952642917633057,
"kl": 0.00986480712890625,
"learning_rate": 1e-06,
"loss": -0.0065,
"num_tokens": 1430304.0,
"reward": -8.034423828125,
"reward_std": 2.978208541870117,
"rewards/rm_reward_func/mean": -8.034423828125,
"rewards/rm_reward_func/std": 7.964836120605469,
"step": 113
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 454.0,
"completions/mean_length": 431.5,
"completions/mean_terminated_length": 297.3333435058594,
"completions/min_length": 93.0,
"completions/min_terminated_length": 93.0,
"epoch": 0.0912,
"grad_norm": 3.2031984329223633,
"kl": 0.012176513671875,
"learning_rate": 1e-06,
"loss": -0.0707,
"num_tokens": 1446624.0,
"reward": -8.611083984375,
"reward_std": 3.2820539474487305,
"rewards/rm_reward_func/mean": -8.611083984375,
"rewards/rm_reward_func/std": 6.850192070007324,
"step": 114
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 437.0,
"completions/mean_length": 244.6875,
"completions/mean_terminated_length": 104.66667175292969,
"completions/min_length": 33.0,
"completions/min_terminated_length": 33.0,
"epoch": 0.092,
"grad_norm": 17.274948120117188,
"kl": 0.0135955810546875,
"learning_rate": 1e-06,
"loss": 0.3859,
"num_tokens": 1461166.0,
"reward": -5.8297882080078125,
"reward_std": 4.278718948364258,
"rewards/rm_reward_func/mean": -5.8297882080078125,
"rewards/rm_reward_func/std": 5.697154521942139,
"step": 115
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 486.0,
"completions/mean_length": 312.5,
"completions/mean_terminated_length": 221.8181915283203,
"completions/min_length": 46.0,
"completions/min_terminated_length": 46.0,
"epoch": 0.0928,
"grad_norm": 3.2622101306915283,
"kl": 0.01324462890625,
"learning_rate": 1e-06,
"loss": -0.178,
"num_tokens": 1474222.0,
"reward": -2.70849609375,
"reward_std": 5.48113489151001,
"rewards/rm_reward_func/mean": -2.70849609375,
"rewards/rm_reward_func/std": 13.66856575012207,
"step": 116
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 494.0,
"completions/mean_length": 210.09375,
"completions/mean_terminated_length": 140.42308044433594,
"completions/min_length": 44.0,
"completions/min_terminated_length": 44.0,
"epoch": 0.0936,
"grad_norm": 12.358490943908691,
"kl": 0.02017974853515625,
"learning_rate": 1e-06,
"loss": 0.0062,
"num_tokens": 1487289.0,
"reward": -5.93212890625,
"reward_std": 5.732447147369385,
"rewards/rm_reward_func/mean": -5.93212890625,
"rewards/rm_reward_func/std": 10.107297897338867,
"step": 117
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 479.0,
"completions/mean_length": 307.0,
"completions/mean_terminated_length": 184.0,
"completions/min_length": 20.0,
"completions/min_terminated_length": 20.0,
"epoch": 0.0944,
"grad_norm": 7.494678974151611,
"kl": 0.012603759765625,
"learning_rate": 1e-06,
"loss": -0.1111,
"num_tokens": 1501401.0,
"reward": -0.51397705078125,
"reward_std": 4.986804008483887,
"rewards/rm_reward_func/mean": -0.51397705078125,
"rewards/rm_reward_func/std": 6.126167297363281,
"step": 118
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 221.59375,
"completions/mean_terminated_length": 202.23333740234375,
"completions/min_length": 39.0,
"completions/min_terminated_length": 39.0,
"epoch": 0.0952,
"grad_norm": 5.377990245819092,
"kl": 0.0181732177734375,
"learning_rate": 1e-06,
"loss": -0.2771,
"num_tokens": 1512988.0,
"reward": -0.9085693359375,
"reward_std": 4.9990386962890625,
"rewards/rm_reward_func/mean": -0.9085693359375,
"rewards/rm_reward_func/std": 6.47190523147583,
"step": 119
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.4375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 467.0,
"completions/mean_length": 307.78125,
"completions/mean_terminated_length": 148.94444274902344,
"completions/min_length": 49.0,
"completions/min_terminated_length": 49.0,
"epoch": 0.096,
"grad_norm": 5.832651138305664,
"kl": 0.01479339599609375,
"learning_rate": 1e-06,
"loss": 0.0782,
"num_tokens": 1530557.0,
"reward": -7.892578125,
"reward_std": 3.8970372676849365,
"rewards/rm_reward_func/mean": -7.892578125,
"rewards/rm_reward_func/std": 6.989450454711914,
"step": 120
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 454.0,
"completions/mean_length": 275.4375,
"completions/mean_terminated_length": 220.84616088867188,
"completions/min_length": 49.0,
"completions/min_terminated_length": 49.0,
"epoch": 0.0968,
"grad_norm": 2.5381200313568115,
"kl": 0.0143890380859375,
"learning_rate": 1e-06,
"loss": -0.1052,
"num_tokens": 1542539.0,
"reward": -1.7589111328125,
"reward_std": 3.9134409427642822,
"rewards/rm_reward_func/mean": -1.7589111328125,
"rewards/rm_reward_func/std": 9.491743087768555,
"step": 121
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.40625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 427.0,
"completions/mean_length": 336.03125,
"completions/mean_terminated_length": 215.63157653808594,
"completions/min_length": 31.0,
"completions/min_terminated_length": 31.0,
"epoch": 0.0976,
"grad_norm": 4.144689559936523,
"kl": 0.014801025390625,
"learning_rate": 1e-06,
"loss": 0.0091,
"num_tokens": 1555908.0,
"reward": -3.1043167114257812,
"reward_std": 3.349581718444824,
"rewards/rm_reward_func/mean": -3.1043167114257812,
"rewards/rm_reward_func/std": 6.525447368621826,
"step": 122
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 281.1875,
"completions/mean_terminated_length": 257.3103332519531,
"completions/min_length": 55.0,
"completions/min_terminated_length": 55.0,
"epoch": 0.0984,
"grad_norm": 10.447542190551758,
"kl": 0.02520751953125,
"learning_rate": 1e-06,
"loss": 0.0106,
"num_tokens": 1568394.0,
"reward": 2.2360823154449463,
"reward_std": 4.762975692749023,
"rewards/rm_reward_func/mean": 2.2360823154449463,
"rewards/rm_reward_func/std": 9.465394973754883,
"step": 123
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 457.0,
"completions/mean_length": 282.90625,
"completions/mean_terminated_length": 240.48147583007812,
"completions/min_length": 23.0,
"completions/min_terminated_length": 23.0,
"epoch": 0.0992,
"grad_norm": 66.1051025390625,
"kl": 0.0311279296875,
"learning_rate": 1e-06,
"loss": 0.0544,
"num_tokens": 1584383.0,
"reward": -4.30712890625,
"reward_std": 7.795045852661133,
"rewards/rm_reward_func/mean": -4.30712890625,
"rewards/rm_reward_func/std": 11.063386917114258,
"step": 124
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 470.0,
"completions/mean_length": 276.84375,
"completions/mean_terminated_length": 252.51724243164062,
"completions/min_length": 47.0,
"completions/min_terminated_length": 47.0,
"epoch": 0.1,
"grad_norm": 3.615330457687378,
"kl": 0.025054931640625,
"learning_rate": 1e-06,
"loss": 0.0065,
"num_tokens": 1595322.0,
"reward": -11.7393798828125,
"reward_std": 2.689405918121338,
"rewards/rm_reward_func/mean": -11.7393798828125,
"rewards/rm_reward_func/std": 13.393643379211426,
"step": 125
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 448.0,
"completions/mean_length": 312.03125,
"completions/mean_terminated_length": 207.2857208251953,
"completions/min_length": 39.0,
"completions/min_terminated_length": 39.0,
"epoch": 0.1008,
"grad_norm": 23.150880813598633,
"kl": 0.0230255126953125,
"learning_rate": 1e-06,
"loss": -0.1352,
"num_tokens": 1609723.0,
"reward": -8.863861083984375,
"reward_std": 6.96150541305542,
"rewards/rm_reward_func/mean": -8.863861083984375,
"rewards/rm_reward_func/std": 9.106093406677246,
"step": 126
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 487.0,
"completions/mean_length": 392.78125,
"completions/mean_terminated_length": 321.25,
"completions/min_length": 55.0,
"completions/min_terminated_length": 55.0,
"epoch": 0.1016,
"grad_norm": 3.9794867038726807,
"kl": 0.011749267578125,
"learning_rate": 1e-06,
"loss": 0.2709,
"num_tokens": 1626636.0,
"reward": -8.689453125,
"reward_std": 4.927318572998047,
"rewards/rm_reward_func/mean": -8.689453125,
"rewards/rm_reward_func/std": 7.077872276306152,
"step": 127
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 485.0,
"completions/mean_length": 326.0,
"completions/mean_terminated_length": 264.0,
"completions/min_length": 32.0,
"completions/min_terminated_length": 32.0,
"epoch": 0.1024,
"grad_norm": 3.827911376953125,
"kl": 0.0189971923828125,
"learning_rate": 1e-06,
"loss": -0.0156,
"num_tokens": 1639084.0,
"reward": -0.710693359375,
"reward_std": 6.628046035766602,
"rewards/rm_reward_func/mean": -0.710693359375,
"rewards/rm_reward_func/std": 9.962486267089844,
"step": 128
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 507.0,
"completions/mean_length": 366.8125,
"completions/mean_terminated_length": 310.0,
"completions/min_length": 53.0,
"completions/min_terminated_length": 53.0,
"epoch": 0.1032,
"grad_norm": 2.512551784515381,
"kl": 0.0133056640625,
"learning_rate": 1e-06,
"loss": 0.1508,
"num_tokens": 1653750.0,
"reward": 2.409698486328125,
"reward_std": 8.444917678833008,
"rewards/rm_reward_func/mean": 2.409698486328125,
"rewards/rm_reward_func/std": 11.581578254699707,
"step": 129
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 466.0,
"completions/mean_length": 309.09375,
"completions/mean_terminated_length": 271.5185241699219,
"completions/min_length": 36.0,
"completions/min_terminated_length": 36.0,
"epoch": 0.104,
"grad_norm": 4.3505144119262695,
"kl": 0.02154541015625,
"learning_rate": 1e-06,
"loss": -0.078,
"num_tokens": 1666745.0,
"reward": -0.958282470703125,
"reward_std": 5.802244663238525,
"rewards/rm_reward_func/mean": -0.958282470703125,
"rewards/rm_reward_func/std": 7.622645378112793,
"step": 130
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 390.0,
"completions/mean_length": 206.59375,
"completions/mean_terminated_length": 162.96429443359375,
"completions/min_length": 45.0,
"completions/min_terminated_length": 45.0,
"epoch": 0.1048,
"grad_norm": 5.6502861976623535,
"kl": 0.01812744140625,
"learning_rate": 1e-06,
"loss": 0.1992,
"num_tokens": 1677804.0,
"reward": -6.44091796875,
"reward_std": 6.14309024810791,
"rewards/rm_reward_func/mean": -6.44091796875,
"rewards/rm_reward_func/std": 10.069220542907715,
"step": 131
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 385.5,
"completions/mean_terminated_length": 309.6000061035156,
"completions/min_length": 92.0,
"completions/min_terminated_length": 92.0,
"epoch": 0.1056,
"grad_norm": 2.8553056716918945,
"kl": 0.02423095703125,
"learning_rate": 1e-06,
"loss": -0.1016,
"num_tokens": 1693340.0,
"reward": 2.6444091796875,
"reward_std": 8.04133415222168,
"rewards/rm_reward_func/mean": 2.6444091796875,
"rewards/rm_reward_func/std": 10.943760871887207,
"step": 132
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 495.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 324.65625,
"completions/mean_terminated_length": 324.65625,
"completions/min_length": 196.0,
"completions/min_terminated_length": 196.0,
"epoch": 0.1064,
"grad_norm": 2.7388484477996826,
"kl": 0.0200958251953125,
"learning_rate": 1e-06,
"loss": 0.0034,
"num_tokens": 1705993.0,
"reward": -2.5941162109375,
"reward_std": 4.211709499359131,
"rewards/rm_reward_func/mean": -2.5941162109375,
"rewards/rm_reward_func/std": 6.4686055183410645,
"step": 133
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 465.0,
"completions/mean_length": 254.71875,
"completions/mean_terminated_length": 207.07408142089844,
"completions/min_length": 43.0,
"completions/min_terminated_length": 43.0,
"epoch": 0.1072,
"grad_norm": 4.826857089996338,
"kl": 0.01715087890625,
"learning_rate": 1e-06,
"loss": 0.2512,
"num_tokens": 1716728.0,
"reward": -6.234710693359375,
"reward_std": 6.557104110717773,
"rewards/rm_reward_func/mean": -6.234710693359375,
"rewards/rm_reward_func/std": 8.982573509216309,
"step": 134
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 395.0,
"completions/mean_length": 201.1875,
"completions/mean_terminated_length": 191.16128540039062,
"completions/min_length": 25.0,
"completions/min_terminated_length": 25.0,
"epoch": 0.108,
"grad_norm": 4.799185752868652,
"kl": 0.0233612060546875,
"learning_rate": 1e-06,
"loss": -0.0819,
"num_tokens": 1726414.0,
"reward": -4.594329833984375,
"reward_std": 5.226075649261475,
"rewards/rm_reward_func/mean": -4.594329833984375,
"rewards/rm_reward_func/std": 6.881073474884033,
"step": 135
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 460.0,
"completions/max_terminated_length": 460.0,
"completions/mean_length": 223.96875,
"completions/mean_terminated_length": 223.96875,
"completions/min_length": 29.0,
"completions/min_terminated_length": 29.0,
"epoch": 0.1088,
"grad_norm": 9.70561408996582,
"kl": 0.025299072265625,
"learning_rate": 1e-06,
"loss": 0.053,
"num_tokens": 1736069.0,
"reward": -0.31732177734375,
"reward_std": 3.850168466567993,
"rewards/rm_reward_func/mean": -0.31732177734375,
"rewards/rm_reward_func/std": 7.10466194152832,
"step": 136
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.40625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 477.0,
"completions/mean_length": 329.125,
"completions/mean_terminated_length": 204.0,
"completions/min_length": 21.0,
"completions/min_terminated_length": 21.0,
"epoch": 0.1096,
"grad_norm": 2.982024669647217,
"kl": 0.0196685791015625,
"learning_rate": 1e-06,
"loss": 0.0927,
"num_tokens": 1749921.0,
"reward": -6.221199035644531,
"reward_std": 5.583423137664795,
"rewards/rm_reward_func/mean": -6.221199035644531,
"rewards/rm_reward_func/std": 7.993007659912109,
"step": 137
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 457.0,
"completions/mean_length": 215.84375,
"completions/mean_terminated_length": 173.5357208251953,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"epoch": 0.1104,
"grad_norm": 7.27449893951416,
"kl": 0.0246734619140625,
"learning_rate": 1e-06,
"loss": 0.0318,
"num_tokens": 1761748.0,
"reward": -2.1171817779541016,
"reward_std": 5.113093376159668,
"rewards/rm_reward_func/mean": -2.1171817779541016,
"rewards/rm_reward_func/std": 6.772409439086914,
"step": 138
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 481.0,
"completions/mean_length": 308.84375,
"completions/mean_terminated_length": 241.125,
"completions/min_length": 36.0,
"completions/min_terminated_length": 36.0,
"epoch": 0.1112,
"grad_norm": 5.053046226501465,
"kl": 0.0249481201171875,
"learning_rate": 1e-06,
"loss": 0.0379,
"num_tokens": 1775439.0,
"reward": -1.70648193359375,
"reward_std": 5.1250834465026855,
"rewards/rm_reward_func/mean": -1.70648193359375,
"rewards/rm_reward_func/std": 6.081603050231934,
"step": 139
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 411.90625,
"completions/mean_terminated_length": 351.8500061035156,
"completions/min_length": 198.0,
"completions/min_terminated_length": 198.0,
"epoch": 0.112,
"grad_norm": 2.3874924182891846,
"kl": 0.0135955810546875,
"learning_rate": 1e-06,
"loss": 0.0458,
"num_tokens": 1792004.0,
"reward": 1.85693359375,
"reward_std": 5.751528263092041,
"rewards/rm_reward_func/mean": 1.85693359375,
"rewards/rm_reward_func/std": 7.656973361968994,
"step": 140
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 502.0,
"completions/mean_length": 251.59375,
"completions/mean_terminated_length": 234.2333526611328,
"completions/min_length": 43.0,
"completions/min_terminated_length": 43.0,
"epoch": 0.1128,
"grad_norm": 3.846666097640991,
"kl": 0.0225677490234375,
"learning_rate": 1e-06,
"loss": 0.103,
"num_tokens": 1802663.0,
"reward": -2.373046875,
"reward_std": 6.561789035797119,
"rewards/rm_reward_func/mean": -2.373046875,
"rewards/rm_reward_func/std": 8.583110809326172,
"step": 141
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 269.46875,
"completions/mean_terminated_length": 213.50001525878906,
"completions/min_length": 62.0,
"completions/min_terminated_length": 62.0,
"epoch": 0.1136,
"grad_norm": 3.5698037147521973,
"kl": 0.03497314453125,
"learning_rate": 1e-06,
"loss": -0.0293,
"num_tokens": 1814766.0,
"reward": 0.341033935546875,
"reward_std": 5.161829948425293,
"rewards/rm_reward_func/mean": 0.341033935546875,
"rewards/rm_reward_func/std": 7.019633769989014,
"step": 142
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.46875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 114.0,
"completions/mean_length": 273.8125,
"completions/mean_terminated_length": 63.64706039428711,
"completions/min_length": 37.0,
"completions/min_terminated_length": 37.0,
"epoch": 0.1144,
"grad_norm": 9.236984252929688,
"kl": 0.0199432373046875,
"learning_rate": 1e-06,
"loss": 0.3644,
"num_tokens": 1827504.0,
"reward": -11.159202575683594,
"reward_std": 6.791610240936279,
"rewards/rm_reward_func/mean": -11.159202575683594,
"rewards/rm_reward_func/std": 8.779293060302734,
"step": 143
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.75,
"completions/max_length": 512.0,
"completions/max_terminated_length": 490.0,
"completions/mean_length": 476.4375,
"completions/mean_terminated_length": 369.75,
"completions/min_length": 124.0,
"completions/min_terminated_length": 124.0,
"epoch": 0.1152,
"grad_norm": 2.5026986598968506,
"kl": 0.0117034912109375,
"learning_rate": 1e-06,
"loss": -0.0064,
"num_tokens": 1847678.0,
"reward": -4.88604736328125,
"reward_std": 4.872498035430908,
"rewards/rm_reward_func/mean": -4.88604736328125,
"rewards/rm_reward_func/std": 9.581089973449707,
"step": 144
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 306.9375,
"completions/mean_terminated_length": 268.96295166015625,
"completions/min_length": 25.0,
"completions/min_terminated_length": 25.0,
"epoch": 0.116,
"grad_norm": 2.9737255573272705,
"kl": 0.03043365478515625,
"learning_rate": 1e-06,
"loss": -0.0469,
"num_tokens": 1863420.0,
"reward": -2.0446929931640625,
"reward_std": 3.8031177520751953,
"rewards/rm_reward_func/mean": -2.0446929931640625,
"rewards/rm_reward_func/std": 5.832355976104736,
"step": 145
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.4375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 483.0,
"completions/mean_length": 349.59375,
"completions/mean_terminated_length": 223.2777862548828,
"completions/min_length": 34.0,
"completions/min_terminated_length": 34.0,
"epoch": 0.1168,
"grad_norm": 4.583868503570557,
"kl": 0.01976776123046875,
"learning_rate": 1e-06,
"loss": 0.0255,
"num_tokens": 1876591.0,
"reward": -1.48095703125,
"reward_std": 4.935769081115723,
"rewards/rm_reward_func/mean": -1.48095703125,
"rewards/rm_reward_func/std": 16.34881019592285,
"step": 146
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.4375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 488.0,
"completions/mean_length": 407.25,
"completions/mean_terminated_length": 325.77777099609375,
"completions/min_length": 42.0,
"completions/min_terminated_length": 42.0,
"epoch": 0.1176,
"grad_norm": 3.11322021484375,
"kl": 0.029876708984375,
"learning_rate": 1e-06,
"loss": -0.1804,
"num_tokens": 1892183.0,
"reward": -4.62982177734375,
"reward_std": 6.981845855712891,
"rewards/rm_reward_func/mean": -4.62982177734375,
"rewards/rm_reward_func/std": 14.073089599609375,
"step": 147
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 490.0,
"completions/mean_length": 291.21875,
"completions/mean_terminated_length": 240.2692413330078,
"completions/min_length": 49.0,
"completions/min_terminated_length": 49.0,
"epoch": 0.1184,
"grad_norm": 3.6348979473114014,
"kl": 0.02703857421875,
"learning_rate": 1e-06,
"loss": -0.0011,
"num_tokens": 1905302.0,
"reward": -3.182891845703125,
"reward_std": 7.350401878356934,
"rewards/rm_reward_func/mean": -3.182891845703125,
"rewards/rm_reward_func/std": 10.496583938598633,
"step": 148
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 206.125,
"completions/mean_terminated_length": 174.48275756835938,
"completions/min_length": 45.0,
"completions/min_terminated_length": 45.0,
"epoch": 0.1192,
"grad_norm": 6.732820987701416,
"kl": 0.0248565673828125,
"learning_rate": 1e-06,
"loss": 0.0798,
"num_tokens": 1915282.0,
"reward": -6.91180419921875,
"reward_std": 5.384483337402344,
"rewards/rm_reward_func/mean": -6.91180419921875,
"rewards/rm_reward_func/std": 8.211865425109863,
"step": 149
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 500.0,
"completions/mean_length": 371.78125,
"completions/mean_terminated_length": 308.04547119140625,
"completions/min_length": 161.0,
"completions/min_terminated_length": 161.0,
"epoch": 0.12,
"grad_norm": 3.473301887512207,
"kl": 0.02490234375,
"learning_rate": 1e-06,
"loss": -0.0287,
"num_tokens": 1929515.0,
"reward": -1.9688720703125,
"reward_std": 5.15925931930542,
"rewards/rm_reward_func/mean": -1.9688720703125,
"rewards/rm_reward_func/std": 8.88673210144043,
"step": 150
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 507.0,
"completions/mean_length": 356.46875,
"completions/mean_terminated_length": 334.25,
"completions/min_length": 127.0,
"completions/min_terminated_length": 127.0,
"epoch": 0.1208,
"grad_norm": 14.675175666809082,
"kl": 0.0251617431640625,
"learning_rate": 1e-06,
"loss": -0.0117,
"num_tokens": 1947602.0,
"reward": 3.12274169921875,
"reward_std": 5.56088399887085,
"rewards/rm_reward_func/mean": 3.12274169921875,
"rewards/rm_reward_func/std": 5.511745452880859,
"step": 151
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 408.75,
"completions/mean_terminated_length": 361.8182067871094,
"completions/min_length": 52.0,
"completions/min_terminated_length": 52.0,
"epoch": 0.1216,
"grad_norm": 17.709932327270508,
"kl": 0.0260009765625,
"learning_rate": 1e-06,
"loss": 0.1719,
"num_tokens": 1964330.0,
"reward": 1.937957763671875,
"reward_std": 5.787369728088379,
"rewards/rm_reward_func/mean": 1.937957763671875,
"rewards/rm_reward_func/std": 7.468286037445068,
"step": 152
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 412.78125,
"completions/mean_terminated_length": 389.8846435546875,
"completions/min_length": 112.0,
"completions/min_terminated_length": 112.0,
"epoch": 0.1224,
"grad_norm": 3.4813315868377686,
"kl": 0.027984619140625,
"learning_rate": 1e-06,
"loss": -0.0986,
"num_tokens": 1981395.0,
"reward": 0.11110877990722656,
"reward_std": 4.848204612731934,
"rewards/rm_reward_func/mean": 0.11110877990722656,
"rewards/rm_reward_func/std": 8.962310791015625,
"step": 153
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.4375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 487.0,
"completions/mean_length": 384.375,
"completions/mean_terminated_length": 285.1111145019531,
"completions/min_length": 118.0,
"completions/min_terminated_length": 118.0,
"epoch": 0.1232,
"grad_norm": 3.2246146202087402,
"kl": 0.022918701171875,
"learning_rate": 1e-06,
"loss": 0.0176,
"num_tokens": 1996591.0,
"reward": 1.41229248046875,
"reward_std": 7.246562957763672,
"rewards/rm_reward_func/mean": 1.41229248046875,
"rewards/rm_reward_func/std": 7.45635461807251,
"step": 154
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 354.34375,
"completions/mean_terminated_length": 325.1481628417969,
"completions/min_length": 69.0,
"completions/min_terminated_length": 69.0,
"epoch": 0.124,
"grad_norm": 3.3056445121765137,
"kl": 0.030120849609375,
"learning_rate": 1e-06,
"loss": -0.1102,
"num_tokens": 2010946.0,
"reward": 10.169677734375,
"reward_std": 4.897927284240723,
"rewards/rm_reward_func/mean": 10.169677734375,
"rewards/rm_reward_func/std": 14.160859107971191,
"step": 155
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 454.0,
"completions/mean_length": 261.9375,
"completions/mean_terminated_length": 215.629638671875,
"completions/min_length": 53.0,
"completions/min_terminated_length": 53.0,
"epoch": 0.1248,
"grad_norm": 27.496681213378906,
"kl": 0.02239990234375,
"learning_rate": 1e-06,
"loss": -0.0082,
"num_tokens": 2022800.0,
"reward": -2.4217529296875,
"reward_std": 6.2300496101379395,
"rewards/rm_reward_func/mean": -2.4217529296875,
"rewards/rm_reward_func/std": 10.626413345336914,
"step": 156
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.46875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 480.0,
"completions/mean_length": 346.875,
"completions/mean_terminated_length": 201.1764678955078,
"completions/min_length": 44.0,
"completions/min_terminated_length": 44.0,
"epoch": 0.1256,
"grad_norm": 4.6230573654174805,
"kl": 0.021440505981445312,
"learning_rate": 1e-06,
"loss": 0.1942,
"num_tokens": 2036684.0,
"reward": -7.7816162109375,
"reward_std": 8.917112350463867,
"rewards/rm_reward_func/mean": -7.7816162109375,
"rewards/rm_reward_func/std": 14.053092956542969,
"step": 157
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 392.03125,
"completions/mean_terminated_length": 369.8148193359375,
"completions/min_length": 65.0,
"completions/min_terminated_length": 65.0,
"epoch": 0.1264,
"grad_norm": 2.703683853149414,
"kl": 0.034088134765625,
"learning_rate": 1e-06,
"loss": -0.0831,
"num_tokens": 2051333.0,
"reward": 0.690460205078125,
"reward_std": 7.271797180175781,
"rewards/rm_reward_func/mean": 0.690460205078125,
"rewards/rm_reward_func/std": 11.116622924804688,
"step": 158
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 486.0,
"completions/mean_length": 276.875,
"completions/mean_terminated_length": 184.86956787109375,
"completions/min_length": 48.0,
"completions/min_terminated_length": 48.0,
"epoch": 0.1272,
"grad_norm": 7.96703577041626,
"kl": 0.01966094970703125,
"learning_rate": 1e-06,
"loss": 0.2494,
"num_tokens": 2063065.0,
"reward": 5.6082763671875,
"reward_std": 6.509998798370361,
"rewards/rm_reward_func/mean": 5.6082763671875,
"rewards/rm_reward_func/std": 8.432168960571289,
"step": 159
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.40625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 497.0,
"completions/mean_length": 380.1875,
"completions/mean_terminated_length": 290.0,
"completions/min_length": 121.0,
"completions/min_terminated_length": 121.0,
"epoch": 0.128,
"grad_norm": 3.4324798583984375,
"kl": 0.033443450927734375,
"learning_rate": 1e-06,
"loss": 0.0201,
"num_tokens": 2077479.0,
"reward": -6.3060302734375,
"reward_std": 4.749573707580566,
"rewards/rm_reward_func/mean": -6.3060302734375,
"rewards/rm_reward_func/std": 13.343415260314941,
"step": 160
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 502.0,
"completions/mean_length": 350.25,
"completions/mean_terminated_length": 296.3333435058594,
"completions/min_length": 22.0,
"completions/min_terminated_length": 22.0,
"epoch": 0.1288,
"grad_norm": 2.8479349613189697,
"kl": 0.023101806640625,
"learning_rate": 1e-06,
"loss": -0.0853,
"num_tokens": 2091847.0,
"reward": -2.8158111572265625,
"reward_std": 3.8607840538024902,
"rewards/rm_reward_func/mean": -2.8158111572265625,
"rewards/rm_reward_func/std": 7.014427185058594,
"step": 161
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 499.0,
"completions/mean_length": 326.90625,
"completions/mean_terminated_length": 242.77273559570312,
"completions/min_length": 19.0,
"completions/min_terminated_length": 19.0,
"epoch": 0.1296,
"grad_norm": 5.42257833480835,
"kl": 0.03411865234375,
"learning_rate": 1e-06,
"loss": 0.025,
"num_tokens": 2105620.0,
"reward": -8.8624267578125,
"reward_std": 3.4918594360351562,
"rewards/rm_reward_func/mean": -8.8624267578125,
"rewards/rm_reward_func/std": 6.637436866760254,
"step": 162
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 482.0,
"completions/mean_length": 197.5,
"completions/mean_terminated_length": 164.96551513671875,
"completions/min_length": 54.0,
"completions/min_terminated_length": 54.0,
"epoch": 0.1304,
"grad_norm": 4.355756759643555,
"kl": 0.037811279296875,
"learning_rate": 1e-06,
"loss": -0.0119,
"num_tokens": 2114116.0,
"reward": -11.6080322265625,
"reward_std": 4.272393226623535,
"rewards/rm_reward_func/mean": -11.6080322265625,
"rewards/rm_reward_func/std": 6.417409896850586,
"step": 163
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 276.53125,
"completions/mean_terminated_length": 210.59999084472656,
"completions/min_length": 47.0,
"completions/min_terminated_length": 47.0,
"epoch": 0.1312,
"grad_norm": 3.7714273929595947,
"kl": 0.04034423828125,
"learning_rate": 1e-06,
"loss": 0.1036,
"num_tokens": 2125205.0,
"reward": -6.275434494018555,
"reward_std": 6.214200973510742,
"rewards/rm_reward_func/mean": -6.275434494018555,
"rewards/rm_reward_func/std": 8.869131088256836,
"step": 164
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 432.0,
"completions/mean_length": 258.21875,
"completions/mean_terminated_length": 241.30001831054688,
"completions/min_length": 42.0,
"completions/min_terminated_length": 42.0,
"epoch": 0.132,
"grad_norm": 7.6665167808532715,
"kl": 0.032684326171875,
"learning_rate": 1e-06,
"loss": -0.0494,
"num_tokens": 2135652.0,
"reward": -7.063232421875,
"reward_std": 4.540660858154297,
"rewards/rm_reward_func/mean": -7.063232421875,
"rewards/rm_reward_func/std": 7.4130377769470215,
"step": 165
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 498.0,
"completions/mean_length": 366.40625,
"completions/mean_terminated_length": 317.875,
"completions/min_length": 139.0,
"completions/min_terminated_length": 139.0,
"epoch": 0.1328,
"grad_norm": 2.8710451126098633,
"kl": 0.0362548828125,
"learning_rate": 1e-06,
"loss": -0.0732,
"num_tokens": 2149737.0,
"reward": -1.594390869140625,
"reward_std": 5.7335591316223145,
"rewards/rm_reward_func/mean": -1.594390869140625,
"rewards/rm_reward_func/std": 10.367554664611816,
"step": 166
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 501.0,
"completions/mean_length": 370.6875,
"completions/mean_terminated_length": 315.39129638671875,
"completions/min_length": 102.0,
"completions/min_terminated_length": 102.0,
"epoch": 0.1336,
"grad_norm": 2.312666416168213,
"kl": 0.01702880859375,
"learning_rate": 1e-06,
"loss": -0.113,
"num_tokens": 2165135.0,
"reward": 1.4799518585205078,
"reward_std": 6.825066566467285,
"rewards/rm_reward_func/mean": 1.4799518585205078,
"rewards/rm_reward_func/std": 11.376241683959961,
"step": 167
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 485.0,
"completions/max_terminated_length": 485.0,
"completions/mean_length": 189.3125,
"completions/mean_terminated_length": 189.3125,
"completions/min_length": 43.0,
"completions/min_terminated_length": 43.0,
"epoch": 0.1344,
"grad_norm": 11.69007396697998,
"kl": 0.06195068359375,
"learning_rate": 1e-06,
"loss": 0.0645,
"num_tokens": 2174529.0,
"reward": 4.023796081542969,
"reward_std": 5.484214782714844,
"rewards/rm_reward_func/mean": 4.023796081542969,
"rewards/rm_reward_func/std": 12.138504028320312,
"step": 168
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 427.0,
"completions/mean_length": 280.0,
"completions/mean_terminated_length": 246.85714721679688,
"completions/min_length": 105.0,
"completions/min_terminated_length": 105.0,
"epoch": 0.1352,
"grad_norm": 3.6376917362213135,
"kl": 0.037750244140625,
"learning_rate": 1e-06,
"loss": -0.0308,
"num_tokens": 2185737.0,
"reward": 3.1946182250976562,
"reward_std": 4.653729438781738,
"rewards/rm_reward_func/mean": 3.1946182250976562,
"rewards/rm_reward_func/std": 8.615864753723145,
"step": 169
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 475.0,
"completions/mean_length": 288.53125,
"completions/mean_terminated_length": 265.4137878417969,
"completions/min_length": 83.0,
"completions/min_terminated_length": 83.0,
"epoch": 0.136,
"grad_norm": 6.496363639831543,
"kl": 0.042877197265625,
"learning_rate": 1e-06,
"loss": -0.1146,
"num_tokens": 2197530.0,
"reward": -3.2793121337890625,
"reward_std": 5.921474456787109,
"rewards/rm_reward_func/mean": -3.2793121337890625,
"rewards/rm_reward_func/std": 9.333773612976074,
"step": 170
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 497.0,
"completions/mean_length": 293.46875,
"completions/mean_terminated_length": 194.13636779785156,
"completions/min_length": 33.0,
"completions/min_terminated_length": 33.0,
"epoch": 0.1368,
"grad_norm": 4.2207770347595215,
"kl": 0.029510498046875,
"learning_rate": 1e-06,
"loss": 0.271,
"num_tokens": 2210929.0,
"reward": -6.6259765625,
"reward_std": 6.164700984954834,
"rewards/rm_reward_func/mean": -6.6259765625,
"rewards/rm_reward_func/std": 8.203378677368164,
"step": 171
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.46875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 425.0,
"completions/mean_terminated_length": 348.23529052734375,
"completions/min_length": 45.0,
"completions/min_terminated_length": 45.0,
"epoch": 0.1376,
"grad_norm": 2.683828592300415,
"kl": 0.02593994140625,
"learning_rate": 1e-06,
"loss": 0.092,
"num_tokens": 2229097.0,
"reward": -2.466796875,
"reward_std": 5.979694843292236,
"rewards/rm_reward_func/mean": -2.466796875,
"rewards/rm_reward_func/std": 21.459537506103516,
"step": 172
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 503.0,
"completions/mean_length": 312.15625,
"completions/mean_terminated_length": 207.4761962890625,
"completions/min_length": 48.0,
"completions/min_terminated_length": 48.0,
"epoch": 0.1384,
"grad_norm": 4.745087146759033,
"kl": 0.027069091796875,
"learning_rate": 1e-06,
"loss": 0.2947,
"num_tokens": 2244958.0,
"reward": 0.202301025390625,
"reward_std": 7.154237270355225,
"rewards/rm_reward_func/mean": 0.202301025390625,
"rewards/rm_reward_func/std": 12.688895225524902,
"step": 173
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.53125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 490.0,
"completions/mean_length": 447.09375,
"completions/mean_terminated_length": 373.5333557128906,
"completions/min_length": 33.0,
"completions/min_terminated_length": 33.0,
"epoch": 0.1392,
"grad_norm": 11.718962669372559,
"kl": 0.0233154296875,
"learning_rate": 1e-06,
"loss": -0.0934,
"num_tokens": 2262017.0,
"reward": 5.55755615234375,
"reward_std": 6.223611354827881,
"rewards/rm_reward_func/mean": 5.55755615234375,
"rewards/rm_reward_func/std": 9.582088470458984,
"step": 174
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 498.0,
"completions/mean_length": 287.4375,
"completions/mean_terminated_length": 280.19354248046875,
"completions/min_length": 64.0,
"completions/min_terminated_length": 64.0,
"epoch": 0.14,
"grad_norm": 4.201067924499512,
"kl": 0.060211181640625,
"learning_rate": 1e-06,
"loss": -0.0737,
"num_tokens": 2275175.0,
"reward": -4.629188537597656,
"reward_std": 5.972632884979248,
"rewards/rm_reward_func/mean": -4.629188537597656,
"rewards/rm_reward_func/std": 9.620247840881348,
"step": 175
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 315.46875,
"completions/mean_terminated_length": 270.1153869628906,
"completions/min_length": 49.0,
"completions/min_terminated_length": 49.0,
"epoch": 0.1408,
"grad_norm": 36.3863639831543,
"kl": 0.034271240234375,
"learning_rate": 1e-06,
"loss": -0.0361,
"num_tokens": 2288718.0,
"reward": 2.839324951171875,
"reward_std": 5.17338752746582,
"rewards/rm_reward_func/mean": 2.839324951171875,
"rewards/rm_reward_func/std": 6.9854583740234375,
"step": 176
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.46875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 459.3125,
"completions/mean_terminated_length": 412.8235168457031,
"completions/min_length": 284.0,
"completions/min_terminated_length": 284.0,
"epoch": 0.1416,
"grad_norm": 2.37556529045105,
"kl": 0.02569580078125,
"learning_rate": 1e-06,
"loss": -0.0249,
"num_tokens": 2305264.0,
"reward": 4.6612548828125,
"reward_std": 4.835301399230957,
"rewards/rm_reward_func/mean": 4.6612548828125,
"rewards/rm_reward_func/std": 6.540696620941162,
"step": 177
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 486.0,
"completions/mean_length": 335.90625,
"completions/mean_terminated_length": 277.2083435058594,
"completions/min_length": 155.0,
"completions/min_terminated_length": 155.0,
"epoch": 0.1424,
"grad_norm": 3.222507953643799,
"kl": 0.023162841796875,
"learning_rate": 1e-06,
"loss": 0.0366,
"num_tokens": 2320461.0,
"reward": 0.611053466796875,
"reward_std": 8.012971878051758,
"rewards/rm_reward_func/mean": 0.611053466796875,
"rewards/rm_reward_func/std": 21.763654708862305,
"step": 178
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 301.0,
"completions/mean_length": 276.0625,
"completions/mean_terminated_length": 197.4166717529297,
"completions/min_length": 69.0,
"completions/min_terminated_length": 69.0,
"epoch": 0.1432,
"grad_norm": 3.5161032676696777,
"kl": 0.0360260009765625,
"learning_rate": 1e-06,
"loss": -0.0888,
"num_tokens": 2331991.0,
"reward": -10.7362060546875,
"reward_std": 3.1551408767700195,
"rewards/rm_reward_func/mean": -10.7362060546875,
"rewards/rm_reward_func/std": 7.49362850189209,
"step": 179
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 487.0,
"completions/mean_length": 237.0625,
"completions/mean_terminated_length": 145.4166717529297,
"completions/min_length": 47.0,
"completions/min_terminated_length": 47.0,
"epoch": 0.144,
"grad_norm": 34.31532287597656,
"kl": 0.02911376953125,
"learning_rate": 1e-06,
"loss": 0.3252,
"num_tokens": 2342809.0,
"reward": -6.35968017578125,
"reward_std": 4.969329833984375,
"rewards/rm_reward_func/mean": -6.35968017578125,
"rewards/rm_reward_func/std": 10.100662231445312,
"step": 180
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 444.0,
"completions/max_terminated_length": 444.0,
"completions/mean_length": 174.9375,
"completions/mean_terminated_length": 174.9375,
"completions/min_length": 37.0,
"completions/min_terminated_length": 37.0,
"epoch": 0.1448,
"grad_norm": 11.092037200927734,
"kl": 0.059326171875,
"learning_rate": 1e-06,
"loss": -0.0305,
"num_tokens": 2351343.0,
"reward": 3.357177734375,
"reward_std": 7.95058012008667,
"rewards/rm_reward_func/mean": 3.357177734375,
"rewards/rm_reward_func/std": 9.181958198547363,
"step": 181
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 473.0,
"completions/mean_length": 221.625,
"completions/mean_terminated_length": 167.8518524169922,
"completions/min_length": 21.0,
"completions/min_terminated_length": 21.0,
"epoch": 0.1456,
"grad_norm": 10.811847686767578,
"kl": 0.04290771484375,
"learning_rate": 1e-06,
"loss": -0.1376,
"num_tokens": 2362931.0,
"reward": -2.74267578125,
"reward_std": 5.7764482498168945,
"rewards/rm_reward_func/mean": -2.74267578125,
"rewards/rm_reward_func/std": 14.296358108520508,
"step": 182
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 501.0,
"completions/mean_length": 350.90625,
"completions/mean_terminated_length": 266.5238037109375,
"completions/min_length": 110.0,
"completions/min_terminated_length": 110.0,
"epoch": 0.1464,
"grad_norm": 11.304616928100586,
"kl": 0.02972412109375,
"learning_rate": 1e-06,
"loss": -0.0109,
"num_tokens": 2376720.0,
"reward": -2.41259765625,
"reward_std": 4.233550071716309,
"rewards/rm_reward_func/mean": -2.41259765625,
"rewards/rm_reward_func/std": 8.830628395080566,
"step": 183
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 483.0,
"completions/mean_length": 388.46875,
"completions/mean_terminated_length": 359.9615478515625,
"completions/min_length": 237.0,
"completions/min_terminated_length": 237.0,
"epoch": 0.1472,
"grad_norm": 17.827436447143555,
"kl": 0.04998779296875,
"learning_rate": 1e-06,
"loss": -0.0211,
"num_tokens": 2391807.0,
"reward": 5.77008056640625,
"reward_std": 6.17415189743042,
"rewards/rm_reward_func/mean": 5.77008056640625,
"rewards/rm_reward_func/std": 7.23347806930542,
"step": 184
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 497.0,
"completions/mean_length": 329.5,
"completions/mean_terminated_length": 258.08697509765625,
"completions/min_length": 26.0,
"completions/min_terminated_length": 26.0,
"epoch": 0.148,
"grad_norm": 20.43731117248535,
"kl": 0.037933349609375,
"learning_rate": 1e-06,
"loss": 0.2097,
"num_tokens": 2406519.0,
"reward": -1.623046875,
"reward_std": 10.683743476867676,
"rewards/rm_reward_func/mean": -1.623046875,
"rewards/rm_reward_func/std": 14.652113914489746,
"step": 185
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 499.0,
"completions/mean_length": 397.8125,
"completions/mean_terminated_length": 345.9090881347656,
"completions/min_length": 101.0,
"completions/min_terminated_length": 101.0,
"epoch": 0.1488,
"grad_norm": 3.0334293842315674,
"kl": 0.0233306884765625,
"learning_rate": 1e-06,
"loss": 0.0335,
"num_tokens": 2421993.0,
"reward": 8.384445190429688,
"reward_std": 11.130109786987305,
"rewards/rm_reward_func/mean": 8.384445190429688,
"rewards/rm_reward_func/std": 12.353397369384766,
"step": 186
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.46875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 493.0,
"completions/mean_length": 363.875,
"completions/mean_terminated_length": 233.1764678955078,
"completions/min_length": 49.0,
"completions/min_terminated_length": 49.0,
"epoch": 0.1496,
"grad_norm": 2.8142430782318115,
"kl": 0.03497314453125,
"learning_rate": 1e-06,
"loss": 0.1352,
"num_tokens": 2435917.0,
"reward": -1.4100570678710938,
"reward_std": 7.787389755249023,
"rewards/rm_reward_func/mean": -1.4100570678710938,
"rewards/rm_reward_func/std": 9.113836288452148,
"step": 187
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 447.0,
"completions/mean_length": 266.53125,
"completions/mean_terminated_length": 250.16668701171875,
"completions/min_length": 88.0,
"completions/min_terminated_length": 88.0,
"epoch": 0.1504,
"grad_norm": 8.15317440032959,
"kl": 0.0330810546875,
"learning_rate": 1e-06,
"loss": -0.0168,
"num_tokens": 2446750.0,
"reward": -1.41357421875,
"reward_std": 6.009856224060059,
"rewards/rm_reward_func/mean": -1.41357421875,
"rewards/rm_reward_func/std": 7.865617752075195,
"step": 188
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 505.0,
"completions/mean_length": 282.34375,
"completions/mean_terminated_length": 162.04762268066406,
"completions/min_length": 16.0,
"completions/min_terminated_length": 16.0,
"epoch": 0.1512,
"grad_norm": 86.80154418945312,
"kl": 0.05029296875,
"learning_rate": 1e-06,
"loss": 0.0314,
"num_tokens": 2458649.0,
"reward": -3.796173095703125,
"reward_std": 4.965305328369141,
"rewards/rm_reward_func/mean": -3.796173095703125,
"rewards/rm_reward_func/std": 7.862216949462891,
"step": 189
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 497.0,
"completions/mean_length": 337.875,
"completions/mean_terminated_length": 246.6666717529297,
"completions/min_length": 49.0,
"completions/min_terminated_length": 49.0,
"epoch": 0.152,
"grad_norm": 3.135280132293701,
"kl": 0.037353515625,
"learning_rate": 1e-06,
"loss": -0.1424,
"num_tokens": 2471605.0,
"reward": 4.255180358886719,
"reward_std": 5.963335990905762,
"rewards/rm_reward_func/mean": 4.255180358886719,
"rewards/rm_reward_func/std": 7.225700855255127,
"step": 190
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 496.0,
"completions/mean_length": 327.96875,
"completions/mean_terminated_length": 217.5500030517578,
"completions/min_length": 76.0,
"completions/min_terminated_length": 76.0,
"epoch": 0.1528,
"grad_norm": 3.981858730316162,
"kl": 0.0426025390625,
"learning_rate": 1e-06,
"loss": 0.187,
"num_tokens": 2484356.0,
"reward": -6.214874267578125,
"reward_std": 4.893754482269287,
"rewards/rm_reward_func/mean": -6.214874267578125,
"rewards/rm_reward_func/std": 11.553343772888184,
"step": 191
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 497.0,
"completions/mean_length": 363.5625,
"completions/mean_terminated_length": 305.478271484375,
"completions/min_length": 46.0,
"completions/min_terminated_length": 46.0,
"epoch": 0.1536,
"grad_norm": 5.650016784667969,
"kl": 0.03961181640625,
"learning_rate": 1e-06,
"loss": 0.2227,
"num_tokens": 2499622.0,
"reward": -3.061126708984375,
"reward_std": 7.803011417388916,
"rewards/rm_reward_func/mean": -3.061126708984375,
"rewards/rm_reward_func/std": 11.103745460510254,
"step": 192
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 483.0,
"completions/mean_length": 282.375,
"completions/mean_terminated_length": 239.8518524169922,
"completions/min_length": 69.0,
"completions/min_terminated_length": 69.0,
"epoch": 0.1544,
"grad_norm": 4.78141975402832,
"kl": 0.041748046875,
"learning_rate": 1e-06,
"loss": 0.0997,
"num_tokens": 2511370.0,
"reward": -4.6614990234375,
"reward_std": 3.844748020172119,
"rewards/rm_reward_func/mean": -4.6614990234375,
"rewards/rm_reward_func/std": 6.409038066864014,
"step": 193
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 505.0,
"completions/mean_length": 218.28125,
"completions/mean_terminated_length": 120.375,
"completions/min_length": 45.0,
"completions/min_terminated_length": 45.0,
"epoch": 0.1552,
"grad_norm": 15.153929710388184,
"kl": 0.03350830078125,
"learning_rate": 1e-06,
"loss": 0.1686,
"num_tokens": 2524115.0,
"reward": 1.6598663330078125,
"reward_std": 6.3394975662231445,
"rewards/rm_reward_func/mean": 1.6598663330078125,
"rewards/rm_reward_func/std": 11.99753475189209,
"step": 194
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 268.0,
"completions/mean_length": 158.875,
"completions/mean_terminated_length": 108.42857360839844,
"completions/min_length": 28.0,
"completions/min_terminated_length": 28.0,
"epoch": 0.156,
"grad_norm": 7.861037731170654,
"kl": 0.0665283203125,
"learning_rate": 1e-06,
"loss": -0.0664,
"num_tokens": 2531239.0,
"reward": -6.14581298828125,
"reward_std": 6.690445423126221,
"rewards/rm_reward_func/mean": -6.14581298828125,
"rewards/rm_reward_func/std": 10.1651029586792,
"step": 195
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 329.625,
"completions/mean_terminated_length": 317.4666748046875,
"completions/min_length": 152.0,
"completions/min_terminated_length": 152.0,
"epoch": 0.1568,
"grad_norm": 8.710630416870117,
"kl": 0.04339599609375,
"learning_rate": 1e-06,
"loss": 0.0139,
"num_tokens": 2543811.0,
"reward": 7.0421142578125,
"reward_std": 7.355923175811768,
"rewards/rm_reward_func/mean": 7.0421142578125,
"rewards/rm_reward_func/std": 15.240861892700195,
"step": 196
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 467.0,
"completions/max_terminated_length": 467.0,
"completions/mean_length": 174.8125,
"completions/mean_terminated_length": 174.8125,
"completions/min_length": 42.0,
"completions/min_terminated_length": 42.0,
"epoch": 0.1576,
"grad_norm": 4.834583759307861,
"kl": 0.03778076171875,
"learning_rate": 1e-06,
"loss": 0.0676,
"num_tokens": 2553517.0,
"reward": 0.12786865234375,
"reward_std": 3.288187265396118,
"rewards/rm_reward_func/mean": 0.12786865234375,
"rewards/rm_reward_func/std": 5.167597770690918,
"step": 197
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 491.0,
"completions/mean_length": 301.75,
"completions/mean_terminated_length": 280.0,
"completions/min_length": 48.0,
"completions/min_terminated_length": 48.0,
"epoch": 0.1584,
"grad_norm": 3.8911170959472656,
"kl": 0.037628173828125,
"learning_rate": 1e-06,
"loss": 0.0381,
"num_tokens": 2565581.0,
"reward": 8.29638671875,
"reward_std": 5.519439697265625,
"rewards/rm_reward_func/mean": 8.29638671875,
"rewards/rm_reward_func/std": 12.758336067199707,
"step": 198
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 317.46875,
"completions/mean_terminated_length": 252.625,
"completions/min_length": 23.0,
"completions/min_terminated_length": 23.0,
"epoch": 0.1592,
"grad_norm": 4.481633186340332,
"kl": 0.0205078125,
"learning_rate": 1e-06,
"loss": 0.0399,
"num_tokens": 2578404.0,
"reward": -3.97637939453125,
"reward_std": 6.747166633605957,
"rewards/rm_reward_func/mean": -3.97637939453125,
"rewards/rm_reward_func/std": 13.583218574523926,
"step": 199
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.4375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 503.0,
"completions/mean_length": 334.21875,
"completions/mean_terminated_length": 195.94444274902344,
"completions/min_length": 45.0,
"completions/min_terminated_length": 45.0,
"epoch": 0.16,
"grad_norm": 2.933378219604492,
"kl": 0.0440826416015625,
"learning_rate": 1e-06,
"loss": 0.0151,
"num_tokens": 2591595.0,
"reward": -2.1455078125,
"reward_std": 2.6753859519958496,
"rewards/rm_reward_func/mean": -2.1455078125,
"rewards/rm_reward_func/std": 7.882194519042969,
"step": 200
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 494.0,
"completions/mean_length": 391.9375,
"completions/mean_terminated_length": 358.3199768066406,
"completions/min_length": 25.0,
"completions/min_terminated_length": 25.0,
"epoch": 0.1608,
"grad_norm": 4.800445079803467,
"kl": 0.02825927734375,
"learning_rate": 1e-06,
"loss": 0.0658,
"num_tokens": 2606889.0,
"reward": 2.6326370239257812,
"reward_std": 7.309446334838867,
"rewards/rm_reward_func/mean": 2.6326370239257812,
"rewards/rm_reward_func/std": 9.031261444091797,
"step": 201
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 456.0,
"completions/mean_length": 322.4375,
"completions/mean_terminated_length": 287.3333435058594,
"completions/min_length": 29.0,
"completions/min_terminated_length": 29.0,
"epoch": 0.1616,
"grad_norm": 6.8586320877075195,
"kl": 0.07305908203125,
"learning_rate": 1e-06,
"loss": 0.0351,
"num_tokens": 2623223.0,
"reward": 0.64892578125,
"reward_std": 3.7500052452087402,
"rewards/rm_reward_func/mean": 0.64892578125,
"rewards/rm_reward_func/std": 11.14384937286377,
"step": 202
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 507.0,
"completions/mean_length": 310.0,
"completions/mean_terminated_length": 230.95652770996094,
"completions/min_length": 51.0,
"completions/min_terminated_length": 51.0,
"epoch": 0.1624,
"grad_norm": 4.2218337059021,
"kl": 0.039154052734375,
"learning_rate": 1e-06,
"loss": 0.1215,
"num_tokens": 2638695.0,
"reward": -3.1102294921875,
"reward_std": 10.1006498336792,
"rewards/rm_reward_func/mean": -3.1102294921875,
"rewards/rm_reward_func/std": 10.23782730102539,
"step": 203
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 470.0,
"completions/mean_length": 271.6875,
"completions/mean_terminated_length": 191.58334350585938,
"completions/min_length": 50.0,
"completions/min_terminated_length": 50.0,
"epoch": 0.1632,
"grad_norm": 4.688625335693359,
"kl": 0.0411376953125,
"learning_rate": 1e-06,
"loss": 0.2514,
"num_tokens": 2652157.0,
"reward": 2.2646484375,
"reward_std": 7.363420486450195,
"rewards/rm_reward_func/mean": 2.2646484375,
"rewards/rm_reward_func/std": 11.030182838439941,
"step": 204
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.40625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 473.0,
"completions/mean_length": 306.90625,
"completions/mean_terminated_length": 166.57894897460938,
"completions/min_length": 36.0,
"completions/min_terminated_length": 36.0,
"epoch": 0.164,
"grad_norm": 4.533467769622803,
"kl": 0.02496337890625,
"learning_rate": 1e-06,
"loss": -0.0195,
"num_tokens": 2664242.0,
"reward": -4.059173583984375,
"reward_std": 5.695614337921143,
"rewards/rm_reward_func/mean": -4.059173583984375,
"rewards/rm_reward_func/std": 7.963764190673828,
"step": 205
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 503.0,
"completions/mean_length": 321.875,
"completions/mean_terminated_length": 247.478271484375,
"completions/min_length": 47.0,
"completions/min_terminated_length": 47.0,
"epoch": 0.1648,
"grad_norm": 4.341475009918213,
"kl": 0.0380401611328125,
"learning_rate": 1e-06,
"loss": 0.0705,
"num_tokens": 2677966.0,
"reward": -6.3792724609375,
"reward_std": 8.557640075683594,
"rewards/rm_reward_func/mean": -6.3792724609375,
"rewards/rm_reward_func/std": 10.657180786132812,
"step": 206
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 505.0,
"completions/mean_length": 254.1875,
"completions/mean_terminated_length": 227.51724243164062,
"completions/min_length": 23.0,
"completions/min_terminated_length": 23.0,
"epoch": 0.1656,
"grad_norm": 18.00819206237793,
"kl": 0.02423095703125,
"learning_rate": 1e-06,
"loss": -0.1429,
"num_tokens": 2688732.0,
"reward": -7.324462890625,
"reward_std": 5.273411750793457,
"rewards/rm_reward_func/mean": -7.324462890625,
"rewards/rm_reward_func/std": 13.72179126739502,
"step": 207
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 491.0,
"completions/mean_length": 219.53125,
"completions/mean_terminated_length": 200.03334045410156,
"completions/min_length": 51.0,
"completions/min_terminated_length": 51.0,
"epoch": 0.1664,
"grad_norm": 9.372909545898438,
"kl": 0.04443359375,
"learning_rate": 1e-06,
"loss": 0.1745,
"num_tokens": 2700973.0,
"reward": -3.869039535522461,
"reward_std": 4.182497978210449,
"rewards/rm_reward_func/mean": -3.869039535522461,
"rewards/rm_reward_func/std": 5.315476417541504,
"step": 208
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 381.875,
"completions/mean_terminated_length": 313.71429443359375,
"completions/min_length": 76.0,
"completions/min_terminated_length": 76.0,
"epoch": 0.1672,
"grad_norm": 14.11308765411377,
"kl": 0.0384521484375,
"learning_rate": 1e-06,
"loss": -0.0134,
"num_tokens": 2715521.0,
"reward": 7.0210418701171875,
"reward_std": 9.68466854095459,
"rewards/rm_reward_func/mean": 7.0210418701171875,
"rewards/rm_reward_func/std": 17.186697006225586,
"step": 209
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 448.0,
"completions/mean_length": 259.0,
"completions/mean_terminated_length": 174.6666717529297,
"completions/min_length": 54.0,
"completions/min_terminated_length": 54.0,
"epoch": 0.168,
"grad_norm": 4.891456604003906,
"kl": 0.071197509765625,
"learning_rate": 1e-06,
"loss": -0.0402,
"num_tokens": 2725753.0,
"reward": -1.91473388671875,
"reward_std": 5.077539920806885,
"rewards/rm_reward_func/mean": -1.91473388671875,
"rewards/rm_reward_func/std": 8.266571044921875,
"step": 210
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.59375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 490.0,
"completions/mean_length": 413.21875,
"completions/mean_terminated_length": 268.8461608886719,
"completions/min_length": 59.0,
"completions/min_terminated_length": 59.0,
"epoch": 0.1688,
"grad_norm": 3.1043269634246826,
"kl": 0.01953125,
"learning_rate": 1e-06,
"loss": 0.1178,
"num_tokens": 2742696.0,
"reward": -9.068359375,
"reward_std": 7.59498405456543,
"rewards/rm_reward_func/mean": -9.068359375,
"rewards/rm_reward_func/std": 10.789223670959473,
"step": 211
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.40625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 491.0,
"completions/mean_length": 359.5,
"completions/mean_terminated_length": 255.15789794921875,
"completions/min_length": 78.0,
"completions/min_terminated_length": 78.0,
"epoch": 0.1696,
"grad_norm": 5.556434631347656,
"kl": 0.03485107421875,
"learning_rate": 1e-06,
"loss": 0.0683,
"num_tokens": 2758216.0,
"reward": 2.6542510986328125,
"reward_std": 4.50178337097168,
"rewards/rm_reward_func/mean": 2.6542510986328125,
"rewards/rm_reward_func/std": 7.546914100646973,
"step": 212
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 499.0,
"completions/mean_length": 250.25,
"completions/mean_terminated_length": 223.1724090576172,
"completions/min_length": 47.0,
"completions/min_terminated_length": 47.0,
"epoch": 0.1704,
"grad_norm": 4.8082275390625,
"kl": 0.03082275390625,
"learning_rate": 1e-06,
"loss": 0.0538,
"num_tokens": 2768344.0,
"reward": 0.75689697265625,
"reward_std": 5.467959880828857,
"rewards/rm_reward_func/mean": 0.75689697265625,
"rewards/rm_reward_func/std": 8.289709091186523,
"step": 213
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 473.0,
"completions/mean_length": 377.5625,
"completions/mean_terminated_length": 307.1428527832031,
"completions/min_length": 108.0,
"completions/min_terminated_length": 108.0,
"epoch": 0.1712,
"grad_norm": 3.0155625343322754,
"kl": 0.032958984375,
"learning_rate": 1e-06,
"loss": -0.0965,
"num_tokens": 2782498.0,
"reward": 0.0223388671875,
"reward_std": 4.46353006362915,
"rewards/rm_reward_func/mean": 0.0223388671875,
"rewards/rm_reward_func/std": 11.193925857543945,
"step": 214
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 488.0,
"completions/mean_length": 203.6875,
"completions/mean_terminated_length": 171.79310607910156,
"completions/min_length": 52.0,
"completions/min_terminated_length": 52.0,
"epoch": 0.172,
"grad_norm": 5.511065483093262,
"kl": 0.0542755126953125,
"learning_rate": 1e-06,
"loss": 0.3333,
"num_tokens": 2792208.0,
"reward": -2.86065673828125,
"reward_std": 5.38389778137207,
"rewards/rm_reward_func/mean": -2.86065673828125,
"rewards/rm_reward_func/std": 11.51030158996582,
"step": 215
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 499.0,
"completions/mean_length": 397.84375,
"completions/mean_terminated_length": 329.3500061035156,
"completions/min_length": 58.0,
"completions/min_terminated_length": 58.0,
"epoch": 0.1728,
"grad_norm": 3.1028032302856445,
"kl": 0.040283203125,
"learning_rate": 1e-06,
"loss": -0.0738,
"num_tokens": 2807251.0,
"reward": 5.0045166015625,
"reward_std": 5.947755813598633,
"rewards/rm_reward_func/mean": 5.0045166015625,
"rewards/rm_reward_func/std": 6.689188480377197,
"step": 216
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 473.0,
"completions/mean_length": 267.875,
"completions/mean_terminated_length": 199.51998901367188,
"completions/min_length": 67.0,
"completions/min_terminated_length": 67.0,
"epoch": 0.1736,
"grad_norm": 12.485280990600586,
"kl": 0.043365478515625,
"learning_rate": 1e-06,
"loss": -0.0573,
"num_tokens": 2822775.0,
"reward": -3.88580322265625,
"reward_std": 5.136831283569336,
"rewards/rm_reward_func/mean": -3.88580322265625,
"rewards/rm_reward_func/std": 11.440207481384277,
"step": 217
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 454.0,
"completions/mean_length": 242.875,
"completions/mean_terminated_length": 153.1666717529297,
"completions/min_length": 56.0,
"completions/min_terminated_length": 56.0,
"epoch": 0.1744,
"grad_norm": 8.996060371398926,
"kl": 0.0584716796875,
"learning_rate": 1e-06,
"loss": 0.3665,
"num_tokens": 2833379.0,
"reward": -2.662109375,
"reward_std": 7.626079082489014,
"rewards/rm_reward_func/mean": -2.662109375,
"rewards/rm_reward_func/std": 9.784847259521484,
"step": 218
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.53125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 499.0,
"completions/mean_length": 413.40625,
"completions/mean_terminated_length": 301.66668701171875,
"completions/min_length": 90.0,
"completions/min_terminated_length": 90.0,
"epoch": 0.1752,
"grad_norm": 3.5558414459228516,
"kl": 0.02426910400390625,
"learning_rate": 1e-06,
"loss": 0.0093,
"num_tokens": 2849992.0,
"reward": -1.936492919921875,
"reward_std": 7.031126976013184,
"rewards/rm_reward_func/mean": -1.936492919921875,
"rewards/rm_reward_func/std": 12.036778450012207,
"step": 219
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 357.40625,
"completions/mean_terminated_length": 305.875,
"completions/min_length": 91.0,
"completions/min_terminated_length": 91.0,
"epoch": 0.176,
"grad_norm": 2.7269515991210938,
"kl": 0.047210693359375,
"learning_rate": 1e-06,
"loss": 0.0937,
"num_tokens": 2865101.0,
"reward": 2.5566253662109375,
"reward_std": 5.476223945617676,
"rewards/rm_reward_func/mean": 2.5566253662109375,
"rewards/rm_reward_func/std": 11.695185661315918,
"step": 220
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 289.53125,
"completions/mean_terminated_length": 274.70001220703125,
"completions/min_length": 53.0,
"completions/min_terminated_length": 53.0,
"epoch": 0.1768,
"grad_norm": 5.313303470611572,
"kl": 0.05035400390625,
"learning_rate": 1e-06,
"loss": 0.295,
"num_tokens": 2877494.0,
"reward": 9.74462890625,
"reward_std": 7.827642917633057,
"rewards/rm_reward_func/mean": 9.74462890625,
"rewards/rm_reward_func/std": 13.538431167602539,
"step": 221
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 327.1875,
"completions/mean_terminated_length": 254.86956787109375,
"completions/min_length": 123.0,
"completions/min_terminated_length": 123.0,
"epoch": 0.1776,
"grad_norm": 4.54102087020874,
"kl": 0.0721435546875,
"learning_rate": 1e-06,
"loss": -0.0192,
"num_tokens": 2890372.0,
"reward": 4.88897705078125,
"reward_std": 3.758185386657715,
"rewards/rm_reward_func/mean": 4.88897705078125,
"rewards/rm_reward_func/std": 7.395822525024414,
"step": 222
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 498.0,
"completions/mean_length": 263.65625,
"completions/mean_terminated_length": 237.96551513671875,
"completions/min_length": 36.0,
"completions/min_terminated_length": 36.0,
"epoch": 0.1784,
"grad_norm": 4.374941825866699,
"kl": 0.082183837890625,
"learning_rate": 1e-06,
"loss": 0.0815,
"num_tokens": 2901145.0,
"reward": -4.588043212890625,
"reward_std": 7.087510585784912,
"rewards/rm_reward_func/mean": -4.588043212890625,
"rewards/rm_reward_func/std": 14.971769332885742,
"step": 223
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 441.0,
"completions/mean_length": 228.15625,
"completions/mean_terminated_length": 198.79310607910156,
"completions/min_length": 53.0,
"completions/min_terminated_length": 53.0,
"epoch": 0.1792,
"grad_norm": 3.8707516193389893,
"kl": 0.0631103515625,
"learning_rate": 1e-06,
"loss": -0.0906,
"num_tokens": 2911422.0,
"reward": 0.30474853515625,
"reward_std": 6.519827842712402,
"rewards/rm_reward_func/mean": 0.30474853515625,
"rewards/rm_reward_func/std": 7.540925979614258,
"step": 224
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 325.0,
"completions/mean_length": 119.1875,
"completions/mean_terminated_length": 106.51612854003906,
"completions/min_length": 49.0,
"completions/min_terminated_length": 49.0,
"epoch": 0.18,
"grad_norm": 8.087342262268066,
"kl": 0.083251953125,
"learning_rate": 1e-06,
"loss": 0.2222,
"num_tokens": 2920028.0,
"reward": -4.3948974609375,
"reward_std": 5.361183166503906,
"rewards/rm_reward_func/mean": -4.3948974609375,
"rewards/rm_reward_func/std": 10.915719032287598,
"step": 225
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 403.0,
"completions/mean_length": 262.78125,
"completions/mean_terminated_length": 205.2692413330078,
"completions/min_length": 34.0,
"completions/min_terminated_length": 34.0,
"epoch": 0.1808,
"grad_norm": 5.448842525482178,
"kl": 0.05535888671875,
"learning_rate": 1e-06,
"loss": 0.1111,
"num_tokens": 2930717.0,
"reward": 1.3837890625,
"reward_std": 7.229766845703125,
"rewards/rm_reward_func/mean": 1.3837890625,
"rewards/rm_reward_func/std": 12.284672737121582,
"step": 226
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 490.0,
"completions/mean_length": 274.03125,
"completions/mean_terminated_length": 240.0357208251953,
"completions/min_length": 58.0,
"completions/min_terminated_length": 58.0,
"epoch": 0.1816,
"grad_norm": 5.66386604309082,
"kl": 0.0574951171875,
"learning_rate": 1e-06,
"loss": 0.2228,
"num_tokens": 2942294.0,
"reward": -2.87408447265625,
"reward_std": 8.044215202331543,
"rewards/rm_reward_func/mean": -2.87408447265625,
"rewards/rm_reward_func/std": 10.464455604553223,
"step": 227
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 367.0,
"completions/max_terminated_length": 367.0,
"completions/mean_length": 184.78125,
"completions/mean_terminated_length": 184.78125,
"completions/min_length": 62.0,
"completions/min_terminated_length": 62.0,
"epoch": 0.1824,
"grad_norm": 6.5997467041015625,
"kl": 0.0521240234375,
"learning_rate": 1e-06,
"loss": 0.0741,
"num_tokens": 2953559.0,
"reward": -5.8045654296875,
"reward_std": 4.83258056640625,
"rewards/rm_reward_func/mean": -5.8045654296875,
"rewards/rm_reward_func/std": 9.505318641662598,
"step": 228
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 293.0625,
"completions/mean_terminated_length": 252.51852416992188,
"completions/min_length": 52.0,
"completions/min_terminated_length": 52.0,
"epoch": 0.1832,
"grad_norm": 3.3603475093841553,
"kl": 0.0386962890625,
"learning_rate": 1e-06,
"loss": 0.0101,
"num_tokens": 2966665.0,
"reward": -2.287109375,
"reward_std": 4.542160987854004,
"rewards/rm_reward_func/mean": -2.287109375,
"rewards/rm_reward_func/std": 8.938611030578613,
"step": 229
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 109.0,
"completions/mean_length": 171.875,
"completions/mean_terminated_length": 58.5,
"completions/min_length": 18.0,
"completions/min_terminated_length": 18.0,
"epoch": 0.184,
"grad_norm": 5.757355213165283,
"kl": 0.051544189453125,
"learning_rate": 1e-06,
"loss": -0.0941,
"num_tokens": 2975517.0,
"reward": -1.37554931640625,
"reward_std": 6.1020684242248535,
"rewards/rm_reward_func/mean": -1.37554931640625,
"rewards/rm_reward_func/std": 9.515417098999023,
"step": 230
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 507.0,
"completions/mean_length": 373.875,
"completions/mean_terminated_length": 319.8260803222656,
"completions/min_length": 23.0,
"completions/min_terminated_length": 23.0,
"epoch": 0.1848,
"grad_norm": 2.7484848499298096,
"kl": 0.03887939453125,
"learning_rate": 1e-06,
"loss": -0.047,
"num_tokens": 2989801.0,
"reward": -6.4609375,
"reward_std": 4.946435928344727,
"rewards/rm_reward_func/mean": -6.4609375,
"rewards/rm_reward_func/std": 7.9586381912231445,
"step": 231
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 410.0,
"completions/mean_length": 218.03125,
"completions/mean_terminated_length": 163.59259033203125,
"completions/min_length": 30.0,
"completions/min_terminated_length": 30.0,
"epoch": 0.1856,
"grad_norm": 4.303516864776611,
"kl": 0.0555419921875,
"learning_rate": 1e-06,
"loss": 0.1213,
"num_tokens": 2999034.0,
"reward": -3.778125762939453,
"reward_std": 6.124919891357422,
"rewards/rm_reward_func/mean": -3.778125762939453,
"rewards/rm_reward_func/std": 6.401005744934082,
"step": 232
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 503.0,
"completions/mean_length": 341.71875,
"completions/mean_terminated_length": 324.10345458984375,
"completions/min_length": 222.0,
"completions/min_terminated_length": 222.0,
"epoch": 0.1864,
"grad_norm": 2.5016844272613525,
"kl": 0.0266265869140625,
"learning_rate": 1e-06,
"loss": -0.0613,
"num_tokens": 3012305.0,
"reward": 0.1171875,
"reward_std": 6.036954402923584,
"rewards/rm_reward_func/mean": 0.1171875,
"rewards/rm_reward_func/std": 8.2446870803833,
"step": 233
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 494.0,
"completions/mean_length": 325.0625,
"completions/mean_terminated_length": 240.09091186523438,
"completions/min_length": 63.0,
"completions/min_terminated_length": 63.0,
"epoch": 0.1872,
"grad_norm": 3.126168966293335,
"kl": 0.033935546875,
"learning_rate": 1e-06,
"loss": 0.088,
"num_tokens": 3024995.0,
"reward": -2.0553665161132812,
"reward_std": 4.089500427246094,
"rewards/rm_reward_func/mean": -2.0553665161132812,
"rewards/rm_reward_func/std": 6.073705196380615,
"step": 234
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 348.03125,
"completions/mean_terminated_length": 331.0689697265625,
"completions/min_length": 181.0,
"completions/min_terminated_length": 181.0,
"epoch": 0.188,
"grad_norm": 2.8671622276306152,
"kl": 0.03570556640625,
"learning_rate": 1e-06,
"loss": -0.0666,
"num_tokens": 3040716.0,
"reward": 5.6767578125,
"reward_std": 5.897713661193848,
"rewards/rm_reward_func/mean": 5.6767578125,
"rewards/rm_reward_func/std": 9.497367858886719,
"step": 235
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 481.0,
"completions/mean_length": 178.53125,
"completions/mean_terminated_length": 144.03448486328125,
"completions/min_length": 23.0,
"completions/min_terminated_length": 23.0,
"epoch": 0.1888,
"grad_norm": 6.242638111114502,
"kl": 0.05078125,
"learning_rate": 1e-06,
"loss": 0.0829,
"num_tokens": 3050005.0,
"reward": -10.707763671875,
"reward_std": 5.651987552642822,
"rewards/rm_reward_func/mean": -10.707763671875,
"rewards/rm_reward_func/std": 7.248280048370361,
"step": 236
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 501.0,
"completions/max_terminated_length": 501.0,
"completions/mean_length": 233.0,
"completions/mean_terminated_length": 233.0,
"completions/min_length": 24.0,
"completions/min_terminated_length": 24.0,
"epoch": 0.1896,
"grad_norm": 3.2471871376037598,
"kl": 0.02520751953125,
"learning_rate": 1e-06,
"loss": -0.0609,
"num_tokens": 3060493.0,
"reward": -0.5721435546875,
"reward_std": 6.302328586578369,
"rewards/rm_reward_func/mean": -0.5721435546875,
"rewards/rm_reward_func/std": 10.102531433105469,
"step": 237
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 453.0,
"completions/mean_terminated_length": 354.66668701171875,
"completions/min_length": 224.0,
"completions/min_terminated_length": 224.0,
"epoch": 0.1904,
"grad_norm": 2.5366711616516113,
"kl": 0.02734375,
"learning_rate": 1e-06,
"loss": 0.0203,
"num_tokens": 3077549.0,
"reward": -0.6923828125,
"reward_std": 4.651421546936035,
"rewards/rm_reward_func/mean": -0.6923828125,
"rewards/rm_reward_func/std": 6.494197368621826,
"step": 238
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 480.0,
"completions/mean_length": 312.65625,
"completions/mean_terminated_length": 266.65386962890625,
"completions/min_length": 74.0,
"completions/min_terminated_length": 74.0,
"epoch": 0.1912,
"grad_norm": 3.5506675243377686,
"kl": 0.038726806640625,
"learning_rate": 1e-06,
"loss": -0.0479,
"num_tokens": 3090450.0,
"reward": -0.1416015625,
"reward_std": 5.028796672821045,
"rewards/rm_reward_func/mean": -0.1416015625,
"rewards/rm_reward_func/std": 7.697935581207275,
"step": 239
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 501.0,
"completions/mean_length": 342.75,
"completions/mean_terminated_length": 265.81817626953125,
"completions/min_length": 81.0,
"completions/min_terminated_length": 81.0,
"epoch": 0.192,
"grad_norm": 4.870519161224365,
"kl": 0.02532958984375,
"learning_rate": 1e-06,
"loss": 0.0788,
"num_tokens": 3105794.0,
"reward": -0.5222930908203125,
"reward_std": 7.223459243774414,
"rewards/rm_reward_func/mean": -0.5222930908203125,
"rewards/rm_reward_func/std": 8.785179138183594,
"step": 240
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 459.0,
"completions/mean_length": 328.71875,
"completions/mean_terminated_length": 294.77777099609375,
"completions/min_length": 65.0,
"completions/min_terminated_length": 65.0,
"epoch": 0.1928,
"grad_norm": 2.8499858379364014,
"kl": 0.036407470703125,
"learning_rate": 1e-06,
"loss": -0.0227,
"num_tokens": 3119169.0,
"reward": -0.429351806640625,
"reward_std": 4.58777379989624,
"rewards/rm_reward_func/mean": -0.429351806640625,
"rewards/rm_reward_func/std": 6.2184648513793945,
"step": 241
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.5,
"completions/max_length": 512.0,
"completions/max_terminated_length": 483.0,
"completions/mean_length": 417.875,
"completions/mean_terminated_length": 323.75,
"completions/min_length": 105.0,
"completions/min_terminated_length": 105.0,
"epoch": 0.1936,
"grad_norm": 3.2108466625213623,
"kl": 0.0294189453125,
"learning_rate": 1e-06,
"loss": 0.0375,
"num_tokens": 3137933.0,
"reward": -4.9796905517578125,
"reward_std": 5.5578694343566895,
"rewards/rm_reward_func/mean": -4.9796905517578125,
"rewards/rm_reward_func/std": 14.907642364501953,
"step": 242
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 485.0,
"completions/mean_length": 221.375,
"completions/mean_terminated_length": 167.55555725097656,
"completions/min_length": 51.0,
"completions/min_terminated_length": 51.0,
"epoch": 0.1944,
"grad_norm": 4.164768695831299,
"kl": 0.061614990234375,
"learning_rate": 1e-06,
"loss": -0.1704,
"num_tokens": 3147577.0,
"reward": -3.182586669921875,
"reward_std": 5.680960655212402,
"rewards/rm_reward_func/mean": -3.182586669921875,
"rewards/rm_reward_func/std": 9.807361602783203,
"step": 243
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 475.0,
"completions/mean_length": 308.59375,
"completions/mean_terminated_length": 270.9259338378906,
"completions/min_length": 102.0,
"completions/min_terminated_length": 102.0,
"epoch": 0.1952,
"grad_norm": 4.503950119018555,
"kl": 0.0273284912109375,
"learning_rate": 1e-06,
"loss": -0.1201,
"num_tokens": 3162324.0,
"reward": -5.913942337036133,
"reward_std": 5.996950626373291,
"rewards/rm_reward_func/mean": -5.913942337036133,
"rewards/rm_reward_func/std": 8.33896255493164,
"step": 244
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 261.0,
"completions/mean_terminated_length": 252.90321350097656,
"completions/min_length": 15.0,
"completions/min_terminated_length": 15.0,
"epoch": 0.196,
"grad_norm": 32.10102081298828,
"kl": 0.049560546875,
"learning_rate": 1e-06,
"loss": -0.0312,
"num_tokens": 3176892.0,
"reward": 5.21258544921875,
"reward_std": 7.160707473754883,
"rewards/rm_reward_func/mean": 5.21258544921875,
"rewards/rm_reward_func/std": 9.029263496398926,
"step": 245
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 350.0,
"completions/mean_length": 174.59375,
"completions/mean_terminated_length": 139.6896514892578,
"completions/min_length": 53.0,
"completions/min_terminated_length": 53.0,
"epoch": 0.1968,
"grad_norm": 7.7096686363220215,
"kl": 0.06005859375,
"learning_rate": 1e-06,
"loss": 0.4931,
"num_tokens": 3185263.0,
"reward": -0.90380859375,
"reward_std": 8.060064315795898,
"rewards/rm_reward_func/mean": -0.90380859375,
"rewards/rm_reward_func/std": 9.48470687866211,
"step": 246
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 474.0,
"completions/mean_length": 252.09375,
"completions/mean_terminated_length": 225.20689392089844,
"completions/min_length": 64.0,
"completions/min_terminated_length": 64.0,
"epoch": 0.1976,
"grad_norm": 3.6530377864837646,
"kl": 0.057586669921875,
"learning_rate": 1e-06,
"loss": -0.0903,
"num_tokens": 3196322.0,
"reward": -0.84429931640625,
"reward_std": 3.671651840209961,
"rewards/rm_reward_func/mean": -0.84429931640625,
"rewards/rm_reward_func/std": 5.721739292144775,
"step": 247
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 463.0,
"completions/mean_length": 216.875,
"completions/mean_terminated_length": 148.7692413330078,
"completions/min_length": 61.0,
"completions/min_terminated_length": 61.0,
"epoch": 0.1984,
"grad_norm": 9.012897491455078,
"kl": 0.060882568359375,
"learning_rate": 1e-06,
"loss": -0.0163,
"num_tokens": 3206374.0,
"reward": -1.3134765625,
"reward_std": 1.9922480583190918,
"rewards/rm_reward_func/mean": -1.3134765625,
"rewards/rm_reward_func/std": 9.720147132873535,
"step": 248
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 504.0,
"completions/mean_length": 221.75,
"completions/mean_terminated_length": 202.40000915527344,
"completions/min_length": 50.0,
"completions/min_terminated_length": 50.0,
"epoch": 0.1992,
"grad_norm": 4.076515197753906,
"kl": 0.05389404296875,
"learning_rate": 1e-06,
"loss": 0.0904,
"num_tokens": 3217438.0,
"reward": 5.044475555419922,
"reward_std": 5.064403533935547,
"rewards/rm_reward_func/mean": 5.044475555419922,
"rewards/rm_reward_func/std": 11.514410972595215,
"step": 249
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.4375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 504.0,
"completions/mean_length": 408.25,
"completions/mean_terminated_length": 327.5555725097656,
"completions/min_length": 110.0,
"completions/min_terminated_length": 110.0,
"epoch": 0.2,
"grad_norm": 2.5681052207946777,
"kl": 0.027313232421875,
"learning_rate": 1e-06,
"loss": 0.0815,
"num_tokens": 3233350.0,
"reward": -2.93603515625,
"reward_std": 3.7280993461608887,
"rewards/rm_reward_func/mean": -2.93603515625,
"rewards/rm_reward_func/std": 12.866559028625488,
"step": 250
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 505.0,
"completions/mean_length": 336.5625,
"completions/mean_terminated_length": 267.9130554199219,
"completions/min_length": 65.0,
"completions/min_terminated_length": 65.0,
"epoch": 0.2008,
"grad_norm": 3.0375123023986816,
"kl": 0.045684814453125,
"learning_rate": 1e-06,
"loss": -0.0519,
"num_tokens": 3248000.0,
"reward": -8.931640625,
"reward_std": 7.065740585327148,
"rewards/rm_reward_func/mean": -8.931640625,
"rewards/rm_reward_func/std": 10.82422924041748,
"step": 251
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 238.0,
"completions/mean_terminated_length": 187.25926208496094,
"completions/min_length": 10.0,
"completions/min_terminated_length": 10.0,
"epoch": 0.2016,
"grad_norm": 5.324024677276611,
"kl": 0.057220458984375,
"learning_rate": 1e-06,
"loss": 0.158,
"num_tokens": 3259632.0,
"reward": -5.93310546875,
"reward_std": 4.27932071685791,
"rewards/rm_reward_func/mean": -5.93310546875,
"rewards/rm_reward_func/std": 12.716781616210938,
"step": 252
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 321.0,
"completions/mean_length": 147.09375,
"completions/mean_terminated_length": 122.76667022705078,
"completions/min_length": 16.0,
"completions/min_terminated_length": 16.0,
"epoch": 0.2024,
"grad_norm": 4.747439861297607,
"kl": 0.0233154296875,
"learning_rate": 1e-06,
"loss": -0.0654,
"num_tokens": 3266675.0,
"reward": -6.288330078125,
"reward_std": 5.211904525756836,
"rewards/rm_reward_func/mean": -6.288330078125,
"rewards/rm_reward_func/std": 9.554975509643555,
"step": 253
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 507.0,
"completions/mean_length": 319.34375,
"completions/mean_terminated_length": 283.6666564941406,
"completions/min_length": 42.0,
"completions/min_terminated_length": 42.0,
"epoch": 0.2032,
"grad_norm": 3.346832513809204,
"kl": 0.0314178466796875,
"learning_rate": 1e-06,
"loss": 0.0012,
"num_tokens": 3280838.0,
"reward": -1.9285697937011719,
"reward_std": 6.478157043457031,
"rewards/rm_reward_func/mean": -1.9285697937011719,
"rewards/rm_reward_func/std": 10.842815399169922,
"step": 254
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 365.46875,
"completions/mean_terminated_length": 360.7419128417969,
"completions/min_length": 205.0,
"completions/min_terminated_length": 205.0,
"epoch": 0.204,
"grad_norm": 2.939826011657715,
"kl": 0.04241943359375,
"learning_rate": 1e-06,
"loss": -0.0565,
"num_tokens": 3294621.0,
"reward": 10.9114990234375,
"reward_std": 7.665554523468018,
"rewards/rm_reward_func/mean": 10.9114990234375,
"rewards/rm_reward_func/std": 11.80000114440918,
"step": 255
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 510.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 238.4375,
"completions/mean_terminated_length": 238.4375,
"completions/min_length": 44.0,
"completions/min_terminated_length": 44.0,
"epoch": 0.2048,
"grad_norm": 3.935816764831543,
"kl": 0.0316162109375,
"learning_rate": 1e-06,
"loss": -0.1044,
"num_tokens": 3304547.0,
"reward": -8.1998291015625,
"reward_std": 5.198800086975098,
"rewards/rm_reward_func/mean": -8.1998291015625,
"rewards/rm_reward_func/std": 8.181034088134766,
"step": 256
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 500.0,
"completions/mean_length": 332.09375,
"completions/mean_terminated_length": 313.4827575683594,
"completions/min_length": 68.0,
"completions/min_terminated_length": 68.0,
"epoch": 0.2056,
"grad_norm": 3.129162311553955,
"kl": 0.049652099609375,
"learning_rate": 1e-06,
"loss": 0.0076,
"num_tokens": 3318494.0,
"reward": 5.046142578125,
"reward_std": 4.183533191680908,
"rewards/rm_reward_func/mean": 5.046142578125,
"rewards/rm_reward_func/std": 7.549215793609619,
"step": 257
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 476.0,
"completions/mean_length": 254.0,
"completions/mean_terminated_length": 168.0,
"completions/min_length": 65.0,
"completions/min_terminated_length": 65.0,
"epoch": 0.2064,
"grad_norm": 14.513141632080078,
"kl": 0.0689849853515625,
"learning_rate": 1e-06,
"loss": -0.0949,
"num_tokens": 3331886.0,
"reward": -0.634033203125,
"reward_std": 4.15193510055542,
"rewards/rm_reward_func/mean": -0.634033203125,
"rewards/rm_reward_func/std": 10.710453033447266,
"step": 258
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 484.0,
"completions/mean_length": 345.96875,
"completions/mean_terminated_length": 290.625,
"completions/min_length": 51.0,
"completions/min_terminated_length": 51.0,
"epoch": 0.2072,
"grad_norm": 3.5183660984039307,
"kl": 0.0648193359375,
"learning_rate": 1e-06,
"loss": 0.1997,
"num_tokens": 3345461.0,
"reward": 2.46270751953125,
"reward_std": 8.18614673614502,
"rewards/rm_reward_func/mean": 2.46270751953125,
"rewards/rm_reward_func/std": 9.014900207519531,
"step": 259
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 300.0,
"completions/max_terminated_length": 300.0,
"completions/mean_length": 181.75,
"completions/mean_terminated_length": 181.75,
"completions/min_length": 33.0,
"completions/min_terminated_length": 33.0,
"epoch": 0.208,
"grad_norm": 4.466091632843018,
"kl": 0.07086181640625,
"learning_rate": 1e-06,
"loss": 0.0176,
"num_tokens": 3356853.0,
"reward": -4.78826904296875,
"reward_std": 4.506010055541992,
"rewards/rm_reward_func/mean": -4.78826904296875,
"rewards/rm_reward_func/std": 9.68130111694336,
"step": 260
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 436.0,
"completions/mean_length": 277.53125,
"completions/mean_terminated_length": 185.78260803222656,
"completions/min_length": 51.0,
"completions/min_terminated_length": 51.0,
"epoch": 0.2088,
"grad_norm": 4.487113952636719,
"kl": 0.04511070251464844,
"learning_rate": 1e-06,
"loss": -0.2554,
"num_tokens": 3371294.0,
"reward": -5.722412109375,
"reward_std": 3.237701654434204,
"rewards/rm_reward_func/mean": -5.722412109375,
"rewards/rm_reward_func/std": 7.949067115783691,
"step": 261
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 509.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 186.78125,
"completions/mean_terminated_length": 186.78125,
"completions/min_length": 18.0,
"completions/min_terminated_length": 18.0,
"epoch": 0.2096,
"grad_norm": 5.796142101287842,
"kl": 0.08880615234375,
"learning_rate": 1e-06,
"loss": -0.1566,
"num_tokens": 3379743.0,
"reward": -4.72845458984375,
"reward_std": 7.553511619567871,
"rewards/rm_reward_func/mean": -4.72845458984375,
"rewards/rm_reward_func/std": 11.902534484863281,
"step": 262
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 500.0,
"completions/mean_length": 199.625,
"completions/mean_terminated_length": 141.7777862548828,
"completions/min_length": 22.0,
"completions/min_terminated_length": 22.0,
"epoch": 0.2104,
"grad_norm": 7.342236042022705,
"kl": 0.06292724609375,
"learning_rate": 1e-06,
"loss": -0.1884,
"num_tokens": 3390747.0,
"reward": 8.73486328125,
"reward_std": 5.577756404876709,
"rewards/rm_reward_func/mean": 8.73486328125,
"rewards/rm_reward_func/std": 9.168144226074219,
"step": 263
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 284.9375,
"completions/mean_terminated_length": 242.88888549804688,
"completions/min_length": 28.0,
"completions/min_terminated_length": 28.0,
"epoch": 0.2112,
"grad_norm": 4.4385223388671875,
"kl": 0.0706787109375,
"learning_rate": 1e-06,
"loss": 0.1751,
"num_tokens": 3401873.0,
"reward": 4.10955810546875,
"reward_std": 7.785256862640381,
"rewards/rm_reward_func/mean": 4.10955810546875,
"rewards/rm_reward_func/std": 9.873637199401855,
"step": 264
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 309.5625,
"completions/mean_terminated_length": 217.5454559326172,
"completions/min_length": 46.0,
"completions/min_terminated_length": 46.0,
"epoch": 0.212,
"grad_norm": 4.364110469818115,
"kl": 0.05230712890625,
"learning_rate": 1e-06,
"loss": -0.1079,
"num_tokens": 3414907.0,
"reward": -8.052734375,
"reward_std": 4.879029273986816,
"rewards/rm_reward_func/mean": -8.052734375,
"rewards/rm_reward_func/std": 6.013888835906982,
"step": 265
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 501.0,
"completions/mean_length": 284.5,
"completions/mean_terminated_length": 148.0,
"completions/min_length": 63.0,
"completions/min_terminated_length": 63.0,
"epoch": 0.2128,
"grad_norm": 6.178150653839111,
"kl": 0.0651397705078125,
"learning_rate": 1e-06,
"loss": 0.2324,
"num_tokens": 3426979.0,
"reward": -0.2777862548828125,
"reward_std": 4.882081985473633,
"rewards/rm_reward_func/mean": -0.2777862548828125,
"rewards/rm_reward_func/std": 12.670929908752441,
"step": 266
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 474.0,
"completions/mean_length": 245.78125,
"completions/mean_terminated_length": 171.239990234375,
"completions/min_length": 10.0,
"completions/min_terminated_length": 10.0,
"epoch": 0.2136,
"grad_norm": 7.719261169433594,
"kl": 0.05291748046875,
"learning_rate": 1e-06,
"loss": 0.0783,
"num_tokens": 3437388.0,
"reward": -0.2825927734375,
"reward_std": 5.749314308166504,
"rewards/rm_reward_func/mean": -0.2825927734375,
"rewards/rm_reward_func/std": 7.607658386230469,
"step": 267
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 438.0,
"completions/mean_length": 264.0,
"completions/mean_terminated_length": 228.57144165039062,
"completions/min_length": 84.0,
"completions/min_terminated_length": 84.0,
"epoch": 0.2144,
"grad_norm": 10.929339408874512,
"kl": 0.05267333984375,
"learning_rate": 1e-06,
"loss": -0.0666,
"num_tokens": 3450292.0,
"reward": -1.218994140625,
"reward_std": 3.6973087787628174,
"rewards/rm_reward_func/mean": -1.218994140625,
"rewards/rm_reward_func/std": 10.225122451782227,
"step": 268
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 428.0,
"completions/mean_length": 312.0,
"completions/mean_terminated_length": 245.33334350585938,
"completions/min_length": 104.0,
"completions/min_terminated_length": 104.0,
"epoch": 0.2152,
"grad_norm": 3.8709096908569336,
"kl": 0.03293609619140625,
"learning_rate": 1e-06,
"loss": 0.118,
"num_tokens": 3463604.0,
"reward": -4.028564453125,
"reward_std": 5.055566787719727,
"rewards/rm_reward_func/mean": -4.028564453125,
"rewards/rm_reward_func/std": 13.777055740356445,
"step": 269
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 382.0,
"completions/mean_length": 154.9375,
"completions/mean_terminated_length": 143.4193572998047,
"completions/min_length": 33.0,
"completions/min_terminated_length": 33.0,
"epoch": 0.216,
"grad_norm": 4.488786220550537,
"kl": 0.09271240234375,
"learning_rate": 1e-06,
"loss": -0.0088,
"num_tokens": 3473114.0,
"reward": 0.6698150634765625,
"reward_std": 2.86566162109375,
"rewards/rm_reward_func/mean": 0.6698150634765625,
"rewards/rm_reward_func/std": 4.058006286621094,
"step": 270
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 424.0,
"completions/mean_length": 255.3125,
"completions/mean_terminated_length": 169.75,
"completions/min_length": 60.0,
"completions/min_terminated_length": 60.0,
"epoch": 0.2168,
"grad_norm": 6.03592586517334,
"kl": 0.0821533203125,
"learning_rate": 1e-06,
"loss": -0.0379,
"num_tokens": 3485300.0,
"reward": -5.101806640625,
"reward_std": 2.251300573348999,
"rewards/rm_reward_func/mean": -5.101806640625,
"rewards/rm_reward_func/std": 12.82911491394043,
"step": 271
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 500.0,
"completions/mean_length": 174.53125,
"completions/mean_terminated_length": 163.64515686035156,
"completions/min_length": 62.0,
"completions/min_terminated_length": 62.0,
"epoch": 0.2176,
"grad_norm": 5.9694294929504395,
"kl": 0.07928466796875,
"learning_rate": 1e-06,
"loss": 0.1671,
"num_tokens": 3495549.0,
"reward": 1.263671875,
"reward_std": 7.210736274719238,
"rewards/rm_reward_func/mean": 1.263671875,
"rewards/rm_reward_func/std": 12.690206527709961,
"step": 272
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 504.0,
"completions/mean_length": 275.59375,
"completions/mean_terminated_length": 221.03846740722656,
"completions/min_length": 56.0,
"completions/min_terminated_length": 56.0,
"epoch": 0.2184,
"grad_norm": 7.359996795654297,
"kl": 0.0755615234375,
"learning_rate": 1e-06,
"loss": -0.0083,
"num_tokens": 3510912.0,
"reward": 9.77337646484375,
"reward_std": 7.382805824279785,
"rewards/rm_reward_func/mean": 9.77337646484375,
"rewards/rm_reward_func/std": 12.175554275512695,
"step": 273
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 482.0,
"completions/mean_length": 209.53125,
"completions/mean_terminated_length": 166.32144165039062,
"completions/min_length": 31.0,
"completions/min_terminated_length": 31.0,
"epoch": 0.2192,
"grad_norm": 4.633665561676025,
"kl": 0.06842041015625,
"learning_rate": 1e-06,
"loss": 0.0266,
"num_tokens": 3521105.0,
"reward": -6.2216796875,
"reward_std": 11.70566177368164,
"rewards/rm_reward_func/mean": -6.2216796875,
"rewards/rm_reward_func/std": 14.380496978759766,
"step": 274
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 308.375,
"completions/mean_terminated_length": 251.36000061035156,
"completions/min_length": 65.0,
"completions/min_terminated_length": 65.0,
"epoch": 0.22,
"grad_norm": 9.020089149475098,
"kl": 0.05230712890625,
"learning_rate": 1e-06,
"loss": 0.0549,
"num_tokens": 3534301.0,
"reward": -1.356536865234375,
"reward_std": 6.025744438171387,
"rewards/rm_reward_func/mean": -1.356536865234375,
"rewards/rm_reward_func/std": 11.09672737121582,
"step": 275
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 488.0,
"completions/mean_length": 215.46875,
"completions/mean_terminated_length": 205.90321350097656,
"completions/min_length": 102.0,
"completions/min_terminated_length": 102.0,
"epoch": 0.2208,
"grad_norm": 3.9002256393432617,
"kl": 0.039642333984375,
"learning_rate": 1e-06,
"loss": 0.0256,
"num_tokens": 3544108.0,
"reward": 1.199432373046875,
"reward_std": 4.459294319152832,
"rewards/rm_reward_func/mean": 1.199432373046875,
"rewards/rm_reward_func/std": 5.9614973068237305,
"step": 276
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 499.0,
"completions/mean_length": 409.0,
"completions/mean_terminated_length": 347.20001220703125,
"completions/min_length": 56.0,
"completions/min_terminated_length": 56.0,
"epoch": 0.2216,
"grad_norm": 2.783045768737793,
"kl": 0.045440673828125,
"learning_rate": 1e-06,
"loss": -0.0599,
"num_tokens": 3559324.0,
"reward": 5.524658203125,
"reward_std": 7.193890571594238,
"rewards/rm_reward_func/mean": 5.524658203125,
"rewards/rm_reward_func/std": 16.028470993041992,
"step": 277
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 452.0,
"completions/mean_length": 208.5625,
"completions/mean_terminated_length": 177.1724090576172,
"completions/min_length": 34.0,
"completions/min_terminated_length": 34.0,
"epoch": 0.2224,
"grad_norm": 4.956029415130615,
"kl": 0.06280517578125,
"learning_rate": 1e-06,
"loss": -0.0151,
"num_tokens": 3568326.0,
"reward": -1.5927276611328125,
"reward_std": 3.9969515800476074,
"rewards/rm_reward_func/mean": -1.5927276611328125,
"rewards/rm_reward_func/std": 9.469999313354492,
"step": 278
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 401.0,
"completions/mean_length": 261.375,
"completions/mean_terminated_length": 147.4545440673828,
"completions/min_length": 53.0,
"completions/min_terminated_length": 53.0,
"epoch": 0.2232,
"grad_norm": 3.7816061973571777,
"kl": 0.062408447265625,
"learning_rate": 1e-06,
"loss": -0.2091,
"num_tokens": 3579634.0,
"reward": -7.9661865234375,
"reward_std": 5.55967378616333,
"rewards/rm_reward_func/mean": -7.9661865234375,
"rewards/rm_reward_func/std": 10.989361763000488,
"step": 279
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 475.0,
"completions/mean_length": 314.46875,
"completions/mean_terminated_length": 195.9499969482422,
"completions/min_length": 59.0,
"completions/min_terminated_length": 59.0,
"epoch": 0.224,
"grad_norm": 3.4097347259521484,
"kl": 0.06787109375,
"learning_rate": 1e-06,
"loss": -0.0373,
"num_tokens": 3592225.0,
"reward": -2.29248046875,
"reward_std": 6.151001930236816,
"rewards/rm_reward_func/mean": -2.29248046875,
"rewards/rm_reward_func/std": 10.61819839477539,
"step": 280
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 346.0,
"completions/mean_length": 225.625,
"completions/mean_terminated_length": 159.53846740722656,
"completions/min_length": 65.0,
"completions/min_terminated_length": 65.0,
"epoch": 0.2248,
"grad_norm": 3.473419427871704,
"kl": 0.08502197265625,
"learning_rate": 1e-06,
"loss": -0.0459,
"num_tokens": 3601253.0,
"reward": -2.1300048828125,
"reward_std": 5.648019790649414,
"rewards/rm_reward_func/mean": -2.1300048828125,
"rewards/rm_reward_func/std": 6.02969217300415,
"step": 281
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 490.0,
"completions/mean_length": 227.625,
"completions/mean_terminated_length": 148.0,
"completions/min_length": 59.0,
"completions/min_terminated_length": 59.0,
"epoch": 0.2256,
"grad_norm": 14.872659683227539,
"kl": 0.08856201171875,
"learning_rate": 1e-06,
"loss": -0.037,
"num_tokens": 3615729.0,
"reward": 3.3761444091796875,
"reward_std": 2.2959201335906982,
"rewards/rm_reward_func/mean": 3.3761444091796875,
"rewards/rm_reward_func/std": 5.013526916503906,
"step": 282
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 483.0,
"completions/mean_length": 282.375,
"completions/mean_terminated_length": 239.8518524169922,
"completions/min_length": 73.0,
"completions/min_terminated_length": 73.0,
"epoch": 0.2264,
"grad_norm": 4.269347190856934,
"kl": 0.05474853515625,
"learning_rate": 1e-06,
"loss": -0.0225,
"num_tokens": 3626917.0,
"reward": 5.6048431396484375,
"reward_std": 4.628182411193848,
"rewards/rm_reward_func/mean": 5.6048431396484375,
"rewards/rm_reward_func/std": 8.509807586669922,
"step": 283
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 266.0,
"completions/max_terminated_length": 266.0,
"completions/mean_length": 95.375,
"completions/mean_terminated_length": 95.375,
"completions/min_length": 18.0,
"completions/min_terminated_length": 18.0,
"epoch": 0.2272,
"grad_norm": 7.475558757781982,
"kl": 0.13861083984375,
"learning_rate": 1e-06,
"loss": 0.0557,
"num_tokens": 3633521.0,
"reward": 1.9283256530761719,
"reward_std": 1.3042426109313965,
"rewards/rm_reward_func/mean": 1.9283256530761719,
"rewards/rm_reward_func/std": 3.519742250442505,
"step": 284
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 490.0,
"completions/mean_length": 194.8125,
"completions/mean_terminated_length": 121.61538696289062,
"completions/min_length": 55.0,
"completions/min_terminated_length": 55.0,
"epoch": 0.228,
"grad_norm": 5.4495086669921875,
"kl": 0.1224365234375,
"learning_rate": 1e-06,
"loss": -0.0286,
"num_tokens": 3642939.0,
"reward": 1.588470458984375,
"reward_std": 2.9584450721740723,
"rewards/rm_reward_func/mean": 1.588470458984375,
"rewards/rm_reward_func/std": 8.191596031188965,
"step": 285
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 503.0,
"completions/mean_length": 224.5625,
"completions/mean_terminated_length": 215.29031372070312,
"completions/min_length": 110.0,
"completions/min_terminated_length": 110.0,
"epoch": 0.2288,
"grad_norm": 3.4381351470947266,
"kl": 0.0645751953125,
"learning_rate": 1e-06,
"loss": 0.0433,
"num_tokens": 3653277.0,
"reward": -6.94952392578125,
"reward_std": 3.9081485271453857,
"rewards/rm_reward_func/mean": -6.94952392578125,
"rewards/rm_reward_func/std": 9.789095878601074,
"step": 286
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 277.375,
"completions/mean_terminated_length": 253.10345458984375,
"completions/min_length": 63.0,
"completions/min_terminated_length": 63.0,
"epoch": 0.2296,
"grad_norm": 2.947699785232544,
"kl": 0.07073974609375,
"learning_rate": 1e-06,
"loss": -0.0916,
"num_tokens": 3664473.0,
"reward": -3.729705810546875,
"reward_std": 6.659585952758789,
"rewards/rm_reward_func/mean": -3.729705810546875,
"rewards/rm_reward_func/std": 9.191458702087402,
"step": 287
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 453.0,
"completions/mean_length": 217.15625,
"completions/mean_terminated_length": 207.64515686035156,
"completions/min_length": 63.0,
"completions/min_terminated_length": 63.0,
"epoch": 0.2304,
"grad_norm": 6.4276814460754395,
"kl": 0.10601806640625,
"learning_rate": 1e-06,
"loss": 0.3546,
"num_tokens": 3675502.0,
"reward": 1.8604736328125,
"reward_std": 5.1977338790893555,
"rewards/rm_reward_func/mean": 1.8604736328125,
"rewards/rm_reward_func/std": 6.9971723556518555,
"step": 288
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 461.0,
"completions/mean_length": 304.46875,
"completions/mean_terminated_length": 235.2916717529297,
"completions/min_length": 59.0,
"completions/min_terminated_length": 59.0,
"epoch": 0.2312,
"grad_norm": 3.1732778549194336,
"kl": 0.0634765625,
"learning_rate": 1e-06,
"loss": -0.0507,
"num_tokens": 3688813.0,
"reward": 2.79443359375,
"reward_std": 5.335002899169922,
"rewards/rm_reward_func/mean": 2.79443359375,
"rewards/rm_reward_func/std": 9.30492115020752,
"step": 289
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 501.0,
"completions/max_terminated_length": 501.0,
"completions/mean_length": 171.0625,
"completions/mean_terminated_length": 171.0625,
"completions/min_length": 57.0,
"completions/min_terminated_length": 57.0,
"epoch": 0.232,
"grad_norm": 11.23501968383789,
"kl": 0.07696533203125,
"learning_rate": 1e-06,
"loss": -0.0426,
"num_tokens": 3698575.0,
"reward": 7.9627685546875,
"reward_std": 3.578082323074341,
"rewards/rm_reward_func/mean": 7.9627685546875,
"rewards/rm_reward_func/std": 10.838711738586426,
"step": 290
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 315.625,
"completions/mean_terminated_length": 238.78260803222656,
"completions/min_length": 68.0,
"completions/min_terminated_length": 68.0,
"epoch": 0.2328,
"grad_norm": 5.33342981338501,
"kl": 0.0753173828125,
"learning_rate": 1e-06,
"loss": -0.1215,
"num_tokens": 3710707.0,
"reward": 0.0675048828125,
"reward_std": 6.095547199249268,
"rewards/rm_reward_func/mean": 0.0675048828125,
"rewards/rm_reward_func/std": 8.877041816711426,
"step": 291
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 293.0,
"completions/max_terminated_length": 293.0,
"completions/mean_length": 101.1875,
"completions/mean_terminated_length": 101.1875,
"completions/min_length": 64.0,
"completions/min_terminated_length": 64.0,
"epoch": 0.2336,
"grad_norm": 4.05195951461792,
"kl": 0.118408203125,
"learning_rate": 1e-06,
"loss": -0.0076,
"num_tokens": 3717129.0,
"reward": 2.620485305786133,
"reward_std": 1.9676257371902466,
"rewards/rm_reward_func/mean": 2.620485305786133,
"rewards/rm_reward_func/std": 4.160001754760742,
"step": 292
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 471.0,
"completions/mean_length": 186.375,
"completions/mean_terminated_length": 164.6666717529297,
"completions/min_length": 63.0,
"completions/min_terminated_length": 63.0,
"epoch": 0.2344,
"grad_norm": 2.865370273590088,
"kl": 0.09393310546875,
"learning_rate": 1e-06,
"loss": 0.1097,
"num_tokens": 3725957.0,
"reward": 1.41033935546875,
"reward_std": 3.00130033493042,
"rewards/rm_reward_func/mean": 1.41033935546875,
"rewards/rm_reward_func/std": 9.897608757019043,
"step": 293
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 500.0,
"completions/mean_length": 328.1875,
"completions/mean_terminated_length": 231.90476989746094,
"completions/min_length": 56.0,
"completions/min_terminated_length": 56.0,
"epoch": 0.2352,
"grad_norm": 3.5064573287963867,
"kl": 0.068359375,
"learning_rate": 1e-06,
"loss": -0.1071,
"num_tokens": 3740923.0,
"reward": -3.520263671875,
"reward_std": 12.347634315490723,
"rewards/rm_reward_func/mean": -3.520263671875,
"rewards/rm_reward_func/std": 15.20130729675293,
"step": 294
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 274.03125,
"completions/mean_terminated_length": 194.70834350585938,
"completions/min_length": 65.0,
"completions/min_terminated_length": 65.0,
"epoch": 0.236,
"grad_norm": 5.17435884475708,
"kl": 0.0858154296875,
"learning_rate": 1e-06,
"loss": 0.2784,
"num_tokens": 3751876.0,
"reward": 0.765625,
"reward_std": 6.9244384765625,
"rewards/rm_reward_func/mean": 0.765625,
"rewards/rm_reward_func/std": 14.027642250061035,
"step": 295
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 487.0,
"completions/mean_length": 200.25,
"completions/mean_terminated_length": 155.71429443359375,
"completions/min_length": 33.0,
"completions/min_terminated_length": 33.0,
"epoch": 0.2368,
"grad_norm": 6.697838306427002,
"kl": 0.0771484375,
"learning_rate": 1e-06,
"loss": 0.232,
"num_tokens": 3761100.0,
"reward": 6.63671875,
"reward_std": 5.848393440246582,
"rewards/rm_reward_func/mean": 6.63671875,
"rewards/rm_reward_func/std": 12.636683464050293,
"step": 296
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 504.0,
"completions/mean_length": 239.28125,
"completions/mean_terminated_length": 162.9199981689453,
"completions/min_length": 65.0,
"completions/min_terminated_length": 65.0,
"epoch": 0.2376,
"grad_norm": 2.993525743484497,
"kl": 0.09796142578125,
"learning_rate": 1e-06,
"loss": -0.0469,
"num_tokens": 3773725.0,
"reward": 0.16357421875,
"reward_std": 4.376680850982666,
"rewards/rm_reward_func/mean": 0.16357421875,
"rewards/rm_reward_func/std": 10.634230613708496,
"step": 297
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 492.0,
"completions/mean_length": 293.59375,
"completions/mean_terminated_length": 279.0333557128906,
"completions/min_length": 65.0,
"completions/min_terminated_length": 65.0,
"epoch": 0.2384,
"grad_norm": 7.341770172119141,
"kl": 0.09283447265625,
"learning_rate": 1e-06,
"loss": 0.2676,
"num_tokens": 3786040.0,
"reward": 3.809326171875,
"reward_std": 3.955899238586426,
"rewards/rm_reward_func/mean": 3.809326171875,
"rewards/rm_reward_func/std": 4.149126052856445,
"step": 298
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 502.0,
"completions/mean_length": 362.9375,
"completions/mean_terminated_length": 328.5384826660156,
"completions/min_length": 69.0,
"completions/min_terminated_length": 69.0,
"epoch": 0.2392,
"grad_norm": 2.8502986431121826,
"kl": 0.05450439453125,
"learning_rate": 1e-06,
"loss": -0.0865,
"num_tokens": 3800550.0,
"reward": -2.6076202392578125,
"reward_std": 6.401535987854004,
"rewards/rm_reward_func/mean": -2.6076202392578125,
"rewards/rm_reward_func/std": 12.927457809448242,
"step": 299
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 499.0,
"completions/mean_length": 333.0,
"completions/mean_terminated_length": 299.85186767578125,
"completions/min_length": 63.0,
"completions/min_terminated_length": 63.0,
"epoch": 0.24,
"grad_norm": 3.4198484420776367,
"kl": 0.087646484375,
"learning_rate": 1e-06,
"loss": -0.0911,
"num_tokens": 3813398.0,
"reward": 10.27294921875,
"reward_std": 8.61447525024414,
"rewards/rm_reward_func/mean": 10.27294921875,
"rewards/rm_reward_func/std": 12.064411163330078,
"step": 300
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 315.0,
"completions/mean_length": 226.9375,
"completions/mean_terminated_length": 174.1481475830078,
"completions/min_length": 66.0,
"completions/min_terminated_length": 66.0,
"epoch": 0.2408,
"grad_norm": 20.889190673828125,
"kl": 0.064208984375,
"learning_rate": 1e-06,
"loss": 0.2754,
"num_tokens": 3824964.0,
"reward": 4.1443328857421875,
"reward_std": 4.807868003845215,
"rewards/rm_reward_func/mean": 4.1443328857421875,
"rewards/rm_reward_func/std": 10.203561782836914,
"step": 301
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 494.0,
"completions/mean_length": 355.03125,
"completions/mean_terminated_length": 293.60870361328125,
"completions/min_length": 63.0,
"completions/min_terminated_length": 63.0,
"epoch": 0.2416,
"grad_norm": 3.480130910873413,
"kl": 0.0916748046875,
"learning_rate": 1e-06,
"loss": 0.1964,
"num_tokens": 3839741.0,
"reward": 0.181640625,
"reward_std": 6.749973773956299,
"rewards/rm_reward_func/mean": 0.181640625,
"rewards/rm_reward_func/std": 8.411787986755371,
"step": 302
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 469.0,
"completions/mean_length": 285.96875,
"completions/mean_terminated_length": 167.57142639160156,
"completions/min_length": 64.0,
"completions/min_terminated_length": 64.0,
"epoch": 0.2424,
"grad_norm": 5.95459508895874,
"kl": 0.073211669921875,
"learning_rate": 1e-06,
"loss": 0.0174,
"num_tokens": 3852788.0,
"reward": -7.36474609375,
"reward_std": 2.4573302268981934,
"rewards/rm_reward_func/mean": -7.36474609375,
"rewards/rm_reward_func/std": 10.699016571044922,
"step": 303
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 474.0,
"completions/mean_length": 293.40625,
"completions/mean_terminated_length": 242.9615478515625,
"completions/min_length": 14.0,
"completions/min_terminated_length": 14.0,
"epoch": 0.2432,
"grad_norm": 3.7543785572052,
"kl": 0.0772705078125,
"learning_rate": 1e-06,
"loss": 0.2814,
"num_tokens": 3864457.0,
"reward": 4.44287109375,
"reward_std": 8.432929039001465,
"rewards/rm_reward_func/mean": 4.44287109375,
"rewards/rm_reward_func/std": 20.573001861572266,
"step": 304
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 487.0,
"completions/mean_length": 198.5,
"completions/mean_terminated_length": 188.3870849609375,
"completions/min_length": 55.0,
"completions/min_terminated_length": 55.0,
"epoch": 0.244,
"grad_norm": 3.925734281539917,
"kl": 0.0897216796875,
"learning_rate": 1e-06,
"loss": 0.0502,
"num_tokens": 3874617.0,
"reward": -5.2659912109375,
"reward_std": 3.777980327606201,
"rewards/rm_reward_func/mean": -5.2659912109375,
"rewards/rm_reward_func/std": 7.426268577575684,
"step": 305
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 472.0,
"completions/mean_length": 257.0,
"completions/mean_terminated_length": 248.77418518066406,
"completions/min_length": 128.0,
"completions/min_terminated_length": 128.0,
"epoch": 0.2448,
"grad_norm": 5.314541816711426,
"kl": 0.0880126953125,
"learning_rate": 1e-06,
"loss": -0.1034,
"num_tokens": 3885481.0,
"reward": -1.97320556640625,
"reward_std": 4.8526153564453125,
"rewards/rm_reward_func/mean": -1.97320556640625,
"rewards/rm_reward_func/std": 8.424205780029297,
"step": 306
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 464.0,
"completions/max_terminated_length": 464.0,
"completions/mean_length": 241.5,
"completions/mean_terminated_length": 241.5,
"completions/min_length": 64.0,
"completions/min_terminated_length": 64.0,
"epoch": 0.2456,
"grad_norm": 3.4764163494110107,
"kl": 0.1007080078125,
"learning_rate": 1e-06,
"loss": 0.0276,
"num_tokens": 3895369.0,
"reward": 14.22119140625,
"reward_std": 3.785148859024048,
"rewards/rm_reward_func/mean": 14.22119140625,
"rewards/rm_reward_func/std": 10.576017379760742,
"step": 307
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 371.0,
"completions/mean_length": 297.125,
"completions/mean_terminated_length": 225.5,
"completions/min_length": 73.0,
"completions/min_terminated_length": 73.0,
"epoch": 0.2464,
"grad_norm": 3.061163902282715,
"kl": 0.042938232421875,
"learning_rate": 1e-06,
"loss": 0.0031,
"num_tokens": 3907725.0,
"reward": -2.1278076171875,
"reward_std": 2.4199328422546387,
"rewards/rm_reward_func/mean": -2.1278076171875,
"rewards/rm_reward_func/std": 5.059024333953857,
"step": 308
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 461.0,
"completions/mean_length": 273.6875,
"completions/mean_terminated_length": 266.0,
"completions/min_length": 63.0,
"completions/min_terminated_length": 63.0,
"epoch": 0.2472,
"grad_norm": 3.6154208183288574,
"kl": 0.08709716796875,
"learning_rate": 1e-06,
"loss": -0.0281,
"num_tokens": 3918539.0,
"reward": 5.895008087158203,
"reward_std": 3.205272912979126,
"rewards/rm_reward_func/mean": 5.895008087158203,
"rewards/rm_reward_func/std": 7.15859317779541,
"step": 309
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 410.0,
"completions/mean_length": 283.6875,
"completions/mean_terminated_length": 231.00001525878906,
"completions/min_length": 86.0,
"completions/min_terminated_length": 86.0,
"epoch": 0.248,
"grad_norm": 2.8838815689086914,
"kl": 0.035888671875,
"learning_rate": 1e-06,
"loss": -0.0877,
"num_tokens": 3932985.0,
"reward": -2.539306640625,
"reward_std": 4.815282344818115,
"rewards/rm_reward_func/mean": -2.539306640625,
"rewards/rm_reward_func/std": 8.142019271850586,
"step": 310
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 295.1875,
"completions/mean_terminated_length": 288.19354248046875,
"completions/min_length": 122.0,
"completions/min_terminated_length": 122.0,
"epoch": 0.2488,
"grad_norm": 3.6053807735443115,
"kl": 0.07330322265625,
"learning_rate": 1e-06,
"loss": -0.0069,
"num_tokens": 3944983.0,
"reward": 5.669269561767578,
"reward_std": 5.6011061668396,
"rewards/rm_reward_func/mean": 5.669269561767578,
"rewards/rm_reward_func/std": 10.164783477783203,
"step": 311
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 277.4375,
"completions/mean_terminated_length": 261.8000183105469,
"completions/min_length": 60.0,
"completions/min_terminated_length": 60.0,
"epoch": 0.2496,
"grad_norm": 4.000564098358154,
"kl": 0.091552734375,
"learning_rate": 1e-06,
"loss": 0.0693,
"num_tokens": 3957221.0,
"reward": 6.09881591796875,
"reward_std": 6.87520170211792,
"rewards/rm_reward_func/mean": 6.09881591796875,
"rewards/rm_reward_func/std": 14.669084548950195,
"step": 312
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 439.0,
"completions/mean_length": 151.09375,
"completions/mean_terminated_length": 113.75862121582031,
"completions/min_length": 44.0,
"completions/min_terminated_length": 44.0,
"epoch": 0.2504,
"grad_norm": 3.9963266849517822,
"kl": 0.10784912109375,
"learning_rate": 1e-06,
"loss": -0.0074,
"num_tokens": 3966792.0,
"reward": 2.5313186645507812,
"reward_std": 3.022167682647705,
"rewards/rm_reward_func/mean": 2.5313186645507812,
"rewards/rm_reward_func/std": 6.7323784828186035,
"step": 313
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 505.0,
"completions/mean_length": 353.5,
"completions/mean_terminated_length": 300.66668701171875,
"completions/min_length": 69.0,
"completions/min_terminated_length": 69.0,
"epoch": 0.2512,
"grad_norm": 7.695732116699219,
"kl": 0.05596923828125,
"learning_rate": 1e-06,
"loss": -0.145,
"num_tokens": 3980432.0,
"reward": -0.392333984375,
"reward_std": 5.235438346862793,
"rewards/rm_reward_func/mean": -0.392333984375,
"rewards/rm_reward_func/std": 11.274943351745605,
"step": 314
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 447.0,
"completions/max_terminated_length": 447.0,
"completions/mean_length": 226.90625,
"completions/mean_terminated_length": 226.90625,
"completions/min_length": 65.0,
"completions/min_terminated_length": 65.0,
"epoch": 0.252,
"grad_norm": 19.040664672851562,
"kl": 0.09423828125,
"learning_rate": 1e-06,
"loss": -0.0073,
"num_tokens": 3993069.0,
"reward": 3.00018310546875,
"reward_std": 2.8838400840759277,
"rewards/rm_reward_func/mean": 3.00018310546875,
"rewards/rm_reward_func/std": 4.894301414489746,
"step": 315
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 479.0,
"completions/mean_length": 191.78125,
"completions/mean_terminated_length": 181.4516143798828,
"completions/min_length": 54.0,
"completions/min_terminated_length": 54.0,
"epoch": 0.2528,
"grad_norm": 12.309869766235352,
"kl": 0.119140625,
"learning_rate": 1e-06,
"loss": 0.2599,
"num_tokens": 4004622.0,
"reward": 5.98388671875,
"reward_std": 5.505092144012451,
"rewards/rm_reward_func/mean": 5.98388671875,
"rewards/rm_reward_func/std": 7.081192970275879,
"step": 316
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 498.0,
"completions/mean_length": 239.34375,
"completions/mean_terminated_length": 211.13792419433594,
"completions/min_length": 62.0,
"completions/min_terminated_length": 62.0,
"epoch": 0.2536,
"grad_norm": 5.658858299255371,
"kl": 0.09930419921875,
"learning_rate": 1e-06,
"loss": -0.0048,
"num_tokens": 4015681.0,
"reward": -1.23388671875,
"reward_std": 4.671802520751953,
"rewards/rm_reward_func/mean": -1.23388671875,
"rewards/rm_reward_func/std": 10.551041603088379,
"step": 317
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 503.0,
"completions/mean_length": 251.28125,
"completions/mean_terminated_length": 191.11538696289062,
"completions/min_length": 64.0,
"completions/min_terminated_length": 64.0,
"epoch": 0.2544,
"grad_norm": 3.543397903442383,
"kl": 0.0836181640625,
"learning_rate": 1e-06,
"loss": -0.0158,
"num_tokens": 4026234.0,
"reward": -0.675323486328125,
"reward_std": 2.8310508728027344,
"rewards/rm_reward_func/mean": -0.675323486328125,
"rewards/rm_reward_func/std": 6.245419502258301,
"step": 318
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 178.8125,
"completions/mean_terminated_length": 178.8125,
"completions/min_length": 29.0,
"completions/min_terminated_length": 29.0,
"epoch": 0.2552,
"grad_norm": 4.103335380554199,
"kl": 0.115234375,
"learning_rate": 1e-06,
"loss": 0.049,
"num_tokens": 4035396.0,
"reward": 3.646728515625,
"reward_std": 3.5850768089294434,
"rewards/rm_reward_func/mean": 3.646728515625,
"rewards/rm_reward_func/std": 9.643186569213867,
"step": 319
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 229.0,
"completions/max_terminated_length": 229.0,
"completions/mean_length": 116.25,
"completions/mean_terminated_length": 116.25,
"completions/min_length": 62.0,
"completions/min_terminated_length": 62.0,
"epoch": 0.256,
"grad_norm": 3.8824329376220703,
"kl": 0.14990234375,
"learning_rate": 1e-06,
"loss": -0.0005,
"num_tokens": 4041572.0,
"reward": -1.460205078125,
"reward_std": 2.3325414657592773,
"rewards/rm_reward_func/mean": -1.460205078125,
"rewards/rm_reward_func/std": 8.575539588928223,
"step": 320
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 498.0,
"completions/mean_length": 319.9375,
"completions/mean_terminated_length": 255.9166717529297,
"completions/min_length": 51.0,
"completions/min_terminated_length": 51.0,
"epoch": 0.2568,
"grad_norm": 8.91545295715332,
"kl": 0.07806396484375,
"learning_rate": 1e-06,
"loss": 0.0173,
"num_tokens": 4054298.0,
"reward": 2.38446044921875,
"reward_std": 3.706024169921875,
"rewards/rm_reward_func/mean": 2.38446044921875,
"rewards/rm_reward_func/std": 6.616323471069336,
"step": 321
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 269.375,
"completions/mean_terminated_length": 224.44444274902344,
"completions/min_length": 42.0,
"completions/min_terminated_length": 42.0,
"epoch": 0.2576,
"grad_norm": 4.994580268859863,
"kl": 0.15380859375,
"learning_rate": 1e-06,
"loss": 0.0178,
"num_tokens": 4065390.0,
"reward": 7.4681549072265625,
"reward_std": 3.772608518600464,
"rewards/rm_reward_func/mean": 7.4681549072265625,
"rewards/rm_reward_func/std": 8.73366641998291,
"step": 322
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 502.0,
"completions/mean_length": 306.84375,
"completions/mean_terminated_length": 238.45834350585938,
"completions/min_length": 62.0,
"completions/min_terminated_length": 62.0,
"epoch": 0.2584,
"grad_norm": 6.513478755950928,
"kl": 0.08917236328125,
"learning_rate": 1e-06,
"loss": 0.4623,
"num_tokens": 4079945.0,
"reward": 1.5455856323242188,
"reward_std": 8.046390533447266,
"rewards/rm_reward_func/mean": 1.5455856323242188,
"rewards/rm_reward_func/std": 9.18606185913086,
"step": 323
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 308.5625,
"completions/mean_terminated_length": 295.0000305175781,
"completions/min_length": 52.0,
"completions/min_terminated_length": 52.0,
"epoch": 0.2592,
"grad_norm": 3.7793478965759277,
"kl": 0.08447265625,
"learning_rate": 1e-06,
"loss": -0.0834,
"num_tokens": 4091579.0,
"reward": 2.5294189453125,
"reward_std": 5.412877559661865,
"rewards/rm_reward_func/mean": 2.5294189453125,
"rewards/rm_reward_func/std": 6.231657028198242,
"step": 324
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 500.0,
"completions/mean_length": 264.53125,
"completions/mean_terminated_length": 218.70370483398438,
"completions/min_length": 54.0,
"completions/min_terminated_length": 54.0,
"epoch": 0.26,
"grad_norm": 15.98507022857666,
"kl": 0.10076904296875,
"learning_rate": 1e-06,
"loss": 0.2493,
"num_tokens": 4105692.0,
"reward": 3.5554351806640625,
"reward_std": 5.774120330810547,
"rewards/rm_reward_func/mean": 3.5554351806640625,
"rewards/rm_reward_func/std": 7.647456169128418,
"step": 325
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 501.0,
"completions/mean_length": 323.625,
"completions/mean_terminated_length": 304.137939453125,
"completions/min_length": 64.0,
"completions/min_terminated_length": 64.0,
"epoch": 0.2608,
"grad_norm": 5.084076881408691,
"kl": 0.070556640625,
"learning_rate": 1e-06,
"loss": -0.157,
"num_tokens": 4119744.0,
"reward": 6.6622314453125,
"reward_std": 5.713457107543945,
"rewards/rm_reward_func/mean": 6.6622314453125,
"rewards/rm_reward_func/std": 9.382472038269043,
"step": 326
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 469.0,
"completions/mean_length": 380.6875,
"completions/mean_terminated_length": 301.8999938964844,
"completions/min_length": 63.0,
"completions/min_terminated_length": 63.0,
"epoch": 0.2616,
"grad_norm": 7.414204120635986,
"kl": 0.09124755859375,
"learning_rate": 1e-06,
"loss": -0.1516,
"num_tokens": 4134790.0,
"reward": -4.882232666015625,
"reward_std": 4.5240888595581055,
"rewards/rm_reward_func/mean": -4.882232666015625,
"rewards/rm_reward_func/std": 9.277235984802246,
"step": 327
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 511.0,
"completions/mean_length": 295.09375,
"completions/mean_terminated_length": 254.92593383789062,
"completions/min_length": 18.0,
"completions/min_terminated_length": 18.0,
"epoch": 0.2624,
"grad_norm": 25.824256896972656,
"kl": 0.07318115234375,
"learning_rate": 1e-06,
"loss": 0.0039,
"num_tokens": 4149569.0,
"reward": 5.16973876953125,
"reward_std": 6.8412652015686035,
"rewards/rm_reward_func/mean": 5.16973876953125,
"rewards/rm_reward_func/std": 9.498034477233887,
"step": 328
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 492.0,
"completions/mean_length": 410.5,
"completions/mean_terminated_length": 364.3636474609375,
"completions/min_length": 157.0,
"completions/min_terminated_length": 157.0,
"epoch": 0.2632,
"grad_norm": 2.756131172180176,
"kl": 0.0433349609375,
"learning_rate": 1e-06,
"loss": -0.1067,
"num_tokens": 4164521.0,
"reward": 0.5872802734375,
"reward_std": 5.443880558013916,
"rewards/rm_reward_func/mean": 0.5872802734375,
"rewards/rm_reward_func/std": 6.557459354400635,
"step": 329
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 499.0,
"completions/mean_length": 302.65625,
"completions/mean_terminated_length": 263.8888854980469,
"completions/min_length": 63.0,
"completions/min_terminated_length": 63.0,
"epoch": 0.264,
"grad_norm": 5.988248348236084,
"kl": 0.10546875,
"learning_rate": 1e-06,
"loss": -0.078,
"num_tokens": 4176206.0,
"reward": -2.9931640625,
"reward_std": 5.075013160705566,
"rewards/rm_reward_func/mean": -2.9931640625,
"rewards/rm_reward_func/std": 7.403041839599609,
"step": 330
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 500.0,
"completions/mean_length": 316.4375,
"completions/mean_terminated_length": 227.5454559326172,
"completions/min_length": 84.0,
"completions/min_terminated_length": 84.0,
"epoch": 0.2648,
"grad_norm": 8.037940979003906,
"kl": 0.087493896484375,
"learning_rate": 1e-06,
"loss": -0.0889,
"num_tokens": 4188604.0,
"reward": -0.03961181640625,
"reward_std": 5.210146903991699,
"rewards/rm_reward_func/mean": -0.03961181640625,
"rewards/rm_reward_func/std": 6.811439514160156,
"step": 331
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 352.6875,
"completions/mean_terminated_length": 342.0666809082031,
"completions/min_length": 137.0,
"completions/min_terminated_length": 137.0,
"epoch": 0.2656,
"grad_norm": 3.725177049636841,
"kl": 0.088134765625,
"learning_rate": 1e-06,
"loss": -0.0404,
"num_tokens": 4201738.0,
"reward": 2.9521484375,
"reward_std": 6.857073783874512,
"rewards/rm_reward_func/mean": 2.9521484375,
"rewards/rm_reward_func/std": 7.544101238250732,
"step": 332
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 386.8125,
"completions/mean_terminated_length": 345.0833435058594,
"completions/min_length": 197.0,
"completions/min_terminated_length": 197.0,
"epoch": 0.2664,
"grad_norm": 2.696730613708496,
"kl": 0.084686279296875,
"learning_rate": 1e-06,
"loss": -0.0032,
"num_tokens": 4216772.0,
"reward": 5.23828125,
"reward_std": 5.675407409667969,
"rewards/rm_reward_func/mean": 5.23828125,
"rewards/rm_reward_func/std": 6.8515472412109375,
"step": 333
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 417.0,
"completions/mean_length": 214.78125,
"completions/mean_terminated_length": 205.19354248046875,
"completions/min_length": 31.0,
"completions/min_terminated_length": 31.0,
"epoch": 0.2672,
"grad_norm": 4.98583459854126,
"kl": 0.0863037109375,
"learning_rate": 1e-06,
"loss": -0.0115,
"num_tokens": 4229349.0,
"reward": -2.3466796875,
"reward_std": 5.685585021972656,
"rewards/rm_reward_func/mean": -2.3466796875,
"rewards/rm_reward_func/std": 7.332208633422852,
"step": 334
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 502.0,
"completions/mean_length": 408.78125,
"completions/mean_terminated_length": 236.75,
"completions/min_length": 60.0,
"completions/min_terminated_length": 60.0,
"epoch": 0.268,
"grad_norm": 3.633908987045288,
"kl": 0.0657958984375,
"learning_rate": 1e-06,
"loss": 0.0563,
"num_tokens": 4247326.0,
"reward": 1.4497833251953125,
"reward_std": 5.859368801116943,
"rewards/rm_reward_func/mean": 1.4497833251953125,
"rewards/rm_reward_func/std": 7.945054054260254,
"step": 335
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 398.375,
"completions/mean_terminated_length": 377.3333435058594,
"completions/min_length": 69.0,
"completions/min_terminated_length": 69.0,
"epoch": 0.2688,
"grad_norm": 6.791285514831543,
"kl": 0.0994873046875,
"learning_rate": 1e-06,
"loss": -0.0843,
"num_tokens": 4263082.0,
"reward": 1.28485107421875,
"reward_std": 5.6787285804748535,
"rewards/rm_reward_func/mean": 1.28485107421875,
"rewards/rm_reward_func/std": 12.219388961791992,
"step": 336
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 452.0,
"completions/mean_length": 312.9375,
"completions/mean_terminated_length": 299.66668701171875,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.2696,
"grad_norm": 3.929457902908325,
"kl": 0.0523681640625,
"learning_rate": 1e-06,
"loss": 0.014,
"num_tokens": 4275208.0,
"reward": 7.05194091796875,
"reward_std": 5.769165992736816,
"rewards/rm_reward_func/mean": 7.05194091796875,
"rewards/rm_reward_func/std": 16.998645782470703,
"step": 337
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 456.34375,
"completions/mean_terminated_length": 422.95001220703125,
"completions/min_length": 213.0,
"completions/min_terminated_length": 213.0,
"epoch": 0.2704,
"grad_norm": 4.277588844299316,
"kl": 0.04718017578125,
"learning_rate": 1e-06,
"loss": -0.0212,
"num_tokens": 4293051.0,
"reward": -1.598388671875,
"reward_std": 5.49388313293457,
"rewards/rm_reward_func/mean": -1.598388671875,
"rewards/rm_reward_func/std": 19.015661239624023,
"step": 338
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 488.0,
"completions/mean_length": 259.65625,
"completions/mean_terminated_length": 251.51612854003906,
"completions/min_length": 38.0,
"completions/min_terminated_length": 38.0,
"epoch": 0.2712,
"grad_norm": 26.45867919921875,
"kl": 0.097412109375,
"learning_rate": 1e-06,
"loss": 0.1121,
"num_tokens": 4309008.0,
"reward": 1.324951171875,
"reward_std": 5.575651168823242,
"rewards/rm_reward_func/mean": 1.324951171875,
"rewards/rm_reward_func/std": 11.64552116394043,
"step": 339
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 490.0,
"completions/mean_length": 341.15625,
"completions/mean_terminated_length": 284.2083435058594,
"completions/min_length": 66.0,
"completions/min_terminated_length": 66.0,
"epoch": 0.272,
"grad_norm": 4.051329135894775,
"kl": 0.10674476623535156,
"learning_rate": 1e-06,
"loss": -0.0366,
"num_tokens": 4325749.0,
"reward": 12.0419921875,
"reward_std": 3.590223550796509,
"rewards/rm_reward_func/mean": 12.0419921875,
"rewards/rm_reward_func/std": 17.978811264038086,
"step": 340
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 418.0,
"completions/mean_length": 245.375,
"completions/mean_terminated_length": 217.79310607910156,
"completions/min_length": 38.0,
"completions/min_terminated_length": 38.0,
"epoch": 0.2728,
"grad_norm": 8.272236824035645,
"kl": 0.07733154296875,
"learning_rate": 1e-06,
"loss": 0.3771,
"num_tokens": 4337033.0,
"reward": 0.50048828125,
"reward_std": 5.649558067321777,
"rewards/rm_reward_func/mean": 0.50048828125,
"rewards/rm_reward_func/std": 5.990384101867676,
"step": 341
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 505.0,
"completions/mean_length": 333.78125,
"completions/mean_terminated_length": 264.0434875488281,
"completions/min_length": 29.0,
"completions/min_terminated_length": 29.0,
"epoch": 0.2736,
"grad_norm": 3.4469985961914062,
"kl": 0.06951904296875,
"learning_rate": 1e-06,
"loss": 0.0304,
"num_tokens": 4349922.0,
"reward": 3.2762298583984375,
"reward_std": 8.944576263427734,
"rewards/rm_reward_func/mean": 3.2762298583984375,
"rewards/rm_reward_func/std": 14.431233406066895,
"step": 342
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 140.0,
"completions/max_terminated_length": 140.0,
"completions/mean_length": 68.78125,
"completions/mean_terminated_length": 68.78125,
"completions/min_length": 14.0,
"completions/min_terminated_length": 14.0,
"epoch": 0.2744,
"grad_norm": 12.23254680633545,
"kl": 0.1826171875,
"learning_rate": 1e-06,
"loss": 0.1112,
"num_tokens": 4355779.0,
"reward": -2.1884765625,
"reward_std": 3.757051944732666,
"rewards/rm_reward_func/mean": -2.1884765625,
"rewards/rm_reward_func/std": 9.511407852172852,
"step": 343
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 492.0,
"completions/mean_length": 284.78125,
"completions/mean_terminated_length": 232.34616088867188,
"completions/min_length": 68.0,
"completions/min_terminated_length": 68.0,
"epoch": 0.2752,
"grad_norm": 5.746858596801758,
"kl": 0.10980224609375,
"learning_rate": 1e-06,
"loss": -0.1269,
"num_tokens": 4368244.0,
"reward": 0.2880859375,
"reward_std": 3.1070127487182617,
"rewards/rm_reward_func/mean": 0.2880859375,
"rewards/rm_reward_func/std": 14.436668395996094,
"step": 344
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 388.0,
"completions/max_terminated_length": 388.0,
"completions/mean_length": 158.59375,
"completions/mean_terminated_length": 158.59375,
"completions/min_length": 33.0,
"completions/min_terminated_length": 33.0,
"epoch": 0.276,
"grad_norm": 6.810152053833008,
"kl": 0.12841796875,
"learning_rate": 1e-06,
"loss": -0.1169,
"num_tokens": 4378415.0,
"reward": 2.64385986328125,
"reward_std": 4.024966716766357,
"rewards/rm_reward_func/mean": 2.64385986328125,
"rewards/rm_reward_func/std": 8.710803985595703,
"step": 345
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 486.0,
"completions/mean_length": 300.15625,
"completions/mean_terminated_length": 189.1904754638672,
"completions/min_length": 63.0,
"completions/min_terminated_length": 63.0,
"epoch": 0.2768,
"grad_norm": 4.056140422821045,
"kl": 0.13568115234375,
"learning_rate": 1e-06,
"loss": 0.0213,
"num_tokens": 4390156.0,
"reward": 8.573406219482422,
"reward_std": 4.77297306060791,
"rewards/rm_reward_func/mean": 8.573406219482422,
"rewards/rm_reward_func/std": 5.832368850708008,
"step": 346
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 499.0,
"completions/mean_length": 387.3125,
"completions/mean_terminated_length": 345.75,
"completions/min_length": 182.0,
"completions/min_terminated_length": 182.0,
"epoch": 0.2776,
"grad_norm": 3.6153457164764404,
"kl": 0.0836181640625,
"learning_rate": 1e-06,
"loss": -0.0447,
"num_tokens": 4405446.0,
"reward": 7.1898193359375,
"reward_std": 6.768939971923828,
"rewards/rm_reward_func/mean": 7.1898193359375,
"rewards/rm_reward_func/std": 12.574838638305664,
"step": 347
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 436.0,
"completions/mean_length": 319.53125,
"completions/mean_terminated_length": 204.0500030517578,
"completions/min_length": 18.0,
"completions/min_terminated_length": 18.0,
"epoch": 0.2784,
"grad_norm": 38.247314453125,
"kl": 0.071136474609375,
"learning_rate": 1e-06,
"loss": -0.0091,
"num_tokens": 4419767.0,
"reward": -6.6529541015625,
"reward_std": 5.824492454528809,
"rewards/rm_reward_func/mean": -6.6529541015625,
"rewards/rm_reward_func/std": 8.597187042236328,
"step": 348
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 477.0,
"completions/mean_length": 318.875,
"completions/mean_terminated_length": 298.89654541015625,
"completions/min_length": 68.0,
"completions/min_terminated_length": 68.0,
"epoch": 0.2792,
"grad_norm": 4.058663845062256,
"kl": 0.082733154296875,
"learning_rate": 1e-06,
"loss": -0.0386,
"num_tokens": 4432003.0,
"reward": -0.1033935546875,
"reward_std": 8.34597396850586,
"rewards/rm_reward_func/mean": -0.1033935546875,
"rewards/rm_reward_func/std": 15.697948455810547,
"step": 349
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 368.59375,
"completions/mean_terminated_length": 282.5500183105469,
"completions/min_length": 59.0,
"completions/min_terminated_length": 59.0,
"epoch": 0.28,
"grad_norm": 27.752031326293945,
"kl": 0.07891845703125,
"learning_rate": 1e-06,
"loss": -0.047,
"num_tokens": 4449870.0,
"reward": 0.47021484375,
"reward_std": 6.845122337341309,
"rewards/rm_reward_func/mean": 0.47021484375,
"rewards/rm_reward_func/std": 11.682084083557129,
"step": 350
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 454.0,
"completions/mean_length": 242.75,
"completions/mean_terminated_length": 234.06451416015625,
"completions/min_length": 74.0,
"completions/min_terminated_length": 74.0,
"epoch": 0.2808,
"grad_norm": 4.997708797454834,
"kl": 0.1334228515625,
"learning_rate": 1e-06,
"loss": -0.0635,
"num_tokens": 4462158.0,
"reward": 7.95166015625,
"reward_std": 8.451520919799805,
"rewards/rm_reward_func/mean": 7.95166015625,
"rewards/rm_reward_func/std": 13.134011268615723,
"step": 351
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 504.0,
"completions/mean_length": 266.6875,
"completions/mean_terminated_length": 170.69564819335938,
"completions/min_length": 58.0,
"completions/min_terminated_length": 58.0,
"epoch": 0.2816,
"grad_norm": 11.950886726379395,
"kl": 0.12762451171875,
"learning_rate": 1e-06,
"loss": 0.0629,
"num_tokens": 4473372.0,
"reward": 2.12890625,
"reward_std": 10.61873722076416,
"rewards/rm_reward_func/mean": 2.12890625,
"rewards/rm_reward_func/std": 17.77381706237793,
"step": 352
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 315.84375,
"completions/mean_terminated_length": 260.91998291015625,
"completions/min_length": 68.0,
"completions/min_terminated_length": 68.0,
"epoch": 0.2824,
"grad_norm": 9.539039611816406,
"kl": 0.135009765625,
"learning_rate": 1e-06,
"loss": 0.2418,
"num_tokens": 4486855.0,
"reward": 10.563232421875,
"reward_std": 8.17442512512207,
"rewards/rm_reward_func/mean": 10.563232421875,
"rewards/rm_reward_func/std": 15.37407398223877,
"step": 353
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 501.0,
"completions/mean_length": 388.25,
"completions/mean_terminated_length": 353.6000061035156,
"completions/min_length": 133.0,
"completions/min_terminated_length": 133.0,
"epoch": 0.2832,
"grad_norm": 3.476372003555298,
"kl": 0.1109619140625,
"learning_rate": 1e-06,
"loss": -0.0289,
"num_tokens": 4501519.0,
"reward": 19.090087890625,
"reward_std": 6.380127906799316,
"rewards/rm_reward_func/mean": 19.090087890625,
"rewards/rm_reward_func/std": 11.411051750183105,
"step": 354
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 487.0,
"completions/mean_length": 316.96875,
"completions/mean_terminated_length": 262.3599853515625,
"completions/min_length": 64.0,
"completions/min_terminated_length": 64.0,
"epoch": 0.284,
"grad_norm": 5.603827953338623,
"kl": 0.129150390625,
"learning_rate": 1e-06,
"loss": -0.0998,
"num_tokens": 4514398.0,
"reward": -2.05572509765625,
"reward_std": 5.069465637207031,
"rewards/rm_reward_func/mean": -2.05572509765625,
"rewards/rm_reward_func/std": 7.577589988708496,
"step": 355
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 369.90625,
"completions/mean_terminated_length": 330.1199951171875,
"completions/min_length": 131.0,
"completions/min_terminated_length": 131.0,
"epoch": 0.2848,
"grad_norm": 3.7223684787750244,
"kl": 0.1007080078125,
"learning_rate": 1e-06,
"loss": 0.011,
"num_tokens": 4528347.0,
"reward": 1.4289093017578125,
"reward_std": 5.715549468994141,
"rewards/rm_reward_func/mean": 1.4289093017578125,
"rewards/rm_reward_func/std": 14.303176879882812,
"step": 356
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 505.0,
"completions/mean_length": 335.28125,
"completions/mean_terminated_length": 323.5000305175781,
"completions/min_length": 145.0,
"completions/min_terminated_length": 145.0,
"epoch": 0.2856,
"grad_norm": 4.0376200675964355,
"kl": 0.0750732421875,
"learning_rate": 1e-06,
"loss": -0.0575,
"num_tokens": 4541772.0,
"reward": 6.256103515625,
"reward_std": 8.777978897094727,
"rewards/rm_reward_func/mean": 6.256103515625,
"rewards/rm_reward_func/std": 14.211292266845703,
"step": 357
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 490.0,
"completions/mean_length": 278.75,
"completions/mean_terminated_length": 235.55555725097656,
"completions/min_length": 53.0,
"completions/min_terminated_length": 53.0,
"epoch": 0.2864,
"grad_norm": 4.349524021148682,
"kl": 0.197265625,
"learning_rate": 1e-06,
"loss": 0.0664,
"num_tokens": 4552908.0,
"reward": 5.4974365234375,
"reward_std": 5.0523176193237305,
"rewards/rm_reward_func/mean": 5.4974365234375,
"rewards/rm_reward_func/std": 15.988303184509277,
"step": 358
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 484.0,
"completions/mean_length": 248.75,
"completions/mean_terminated_length": 240.258056640625,
"completions/min_length": 58.0,
"completions/min_terminated_length": 58.0,
"epoch": 0.2872,
"grad_norm": 4.672142028808594,
"kl": 0.200927734375,
"learning_rate": 1e-06,
"loss": 0.0244,
"num_tokens": 4563372.0,
"reward": -1.47747802734375,
"reward_std": 4.878537654876709,
"rewards/rm_reward_func/mean": -1.47747802734375,
"rewards/rm_reward_func/std": 9.86972713470459,
"step": 359
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 502.0,
"completions/mean_length": 277.15625,
"completions/mean_terminated_length": 211.39999389648438,
"completions/min_length": 32.0,
"completions/min_terminated_length": 32.0,
"epoch": 0.288,
"grad_norm": 6.796772480010986,
"kl": 0.0882568359375,
"learning_rate": 1e-06,
"loss": -0.0351,
"num_tokens": 4574377.0,
"reward": 1.7371826171875,
"reward_std": 6.135521411895752,
"rewards/rm_reward_func/mean": 1.7371826171875,
"rewards/rm_reward_func/std": 14.885226249694824,
"step": 360
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 493.0,
"completions/mean_length": 347.6875,
"completions/mean_terminated_length": 336.73333740234375,
"completions/min_length": 66.0,
"completions/min_terminated_length": 66.0,
"epoch": 0.2888,
"grad_norm": 7.4351806640625,
"kl": 0.1314697265625,
"learning_rate": 1e-06,
"loss": -0.1156,
"num_tokens": 4587791.0,
"reward": -0.7554969787597656,
"reward_std": 5.259791851043701,
"rewards/rm_reward_func/mean": -0.7554969787597656,
"rewards/rm_reward_func/std": 12.53715991973877,
"step": 361
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 499.0,
"completions/mean_length": 321.59375,
"completions/mean_terminated_length": 294.39288330078125,
"completions/min_length": 69.0,
"completions/min_terminated_length": 69.0,
"epoch": 0.2896,
"grad_norm": 5.127910614013672,
"kl": 0.08160400390625,
"learning_rate": 1e-06,
"loss": -0.0329,
"num_tokens": 4600466.0,
"reward": 1.8590850830078125,
"reward_std": 7.74429988861084,
"rewards/rm_reward_func/mean": 1.8590850830078125,
"rewards/rm_reward_func/std": 14.955947875976562,
"step": 362
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 460.0,
"completions/mean_length": 169.59375,
"completions/mean_terminated_length": 158.5483856201172,
"completions/min_length": 68.0,
"completions/min_terminated_length": 68.0,
"epoch": 0.2904,
"grad_norm": 4.208690643310547,
"kl": 0.1861572265625,
"learning_rate": 1e-06,
"loss": 0.0933,
"num_tokens": 4609349.0,
"reward": 3.2529525756835938,
"reward_std": 2.592770576477051,
"rewards/rm_reward_func/mean": 3.2529525756835938,
"rewards/rm_reward_func/std": 9.945049285888672,
"step": 363
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 502.0,
"completions/mean_length": 419.96875,
"completions/mean_terminated_length": 364.75,
"completions/min_length": 69.0,
"completions/min_terminated_length": 69.0,
"epoch": 0.2912,
"grad_norm": 3.5521671772003174,
"kl": 0.079864501953125,
"learning_rate": 1e-06,
"loss": -0.1582,
"num_tokens": 4626428.0,
"reward": 2.2109375,
"reward_std": 9.090686798095703,
"rewards/rm_reward_func/mean": 2.2109375,
"rewards/rm_reward_func/std": 24.739288330078125,
"step": 364
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 494.0,
"completions/mean_length": 237.53125,
"completions/mean_terminated_length": 209.13792419433594,
"completions/min_length": 45.0,
"completions/min_terminated_length": 45.0,
"epoch": 0.292,
"grad_norm": 6.295807838439941,
"kl": 0.051605224609375,
"learning_rate": 1e-06,
"loss": 0.0554,
"num_tokens": 4639565.0,
"reward": -10.0103759765625,
"reward_std": 5.414639472961426,
"rewards/rm_reward_func/mean": -10.0103759765625,
"rewards/rm_reward_func/std": 11.643550872802734,
"step": 365
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 502.0,
"completions/mean_length": 335.84375,
"completions/mean_terminated_length": 286.5199890136719,
"completions/min_length": 104.0,
"completions/min_terminated_length": 104.0,
"epoch": 0.2928,
"grad_norm": 4.325115203857422,
"kl": 0.0908203125,
"learning_rate": 1e-06,
"loss": -0.0157,
"num_tokens": 4652608.0,
"reward": 13.1593017578125,
"reward_std": 7.079737663269043,
"rewards/rm_reward_func/mean": 13.1593017578125,
"rewards/rm_reward_func/std": 17.48756980895996,
"step": 366
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 364.25,
"completions/mean_terminated_length": 275.6000061035156,
"completions/min_length": 70.0,
"completions/min_terminated_length": 70.0,
"epoch": 0.2936,
"grad_norm": 4.418249130249023,
"kl": 0.169189453125,
"learning_rate": 1e-06,
"loss": 0.1931,
"num_tokens": 4667984.0,
"reward": 5.2593994140625,
"reward_std": 7.445150375366211,
"rewards/rm_reward_func/mean": 5.2593994140625,
"rewards/rm_reward_func/std": 19.743183135986328,
"step": 367
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 490.0,
"completions/mean_length": 297.8125,
"completions/mean_terminated_length": 258.1481628417969,
"completions/min_length": 69.0,
"completions/min_terminated_length": 69.0,
"epoch": 0.2944,
"grad_norm": 3.9477925300598145,
"kl": 0.167724609375,
"learning_rate": 1e-06,
"loss": -0.0239,
"num_tokens": 4681170.0,
"reward": 7.892822265625,
"reward_std": 4.728097915649414,
"rewards/rm_reward_func/mean": 7.892822265625,
"rewards/rm_reward_func/std": 5.967565536499023,
"step": 368
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 448.0,
"completions/mean_length": 281.28125,
"completions/mean_terminated_length": 238.55555725097656,
"completions/min_length": 67.0,
"completions/min_terminated_length": 67.0,
"epoch": 0.2952,
"grad_norm": 6.2598443031311035,
"kl": 0.108642578125,
"learning_rate": 1e-06,
"loss": -0.2548,
"num_tokens": 4692299.0,
"reward": -7.61279296875,
"reward_std": 4.482115745544434,
"rewards/rm_reward_func/mean": -7.61279296875,
"rewards/rm_reward_func/std": 9.240117073059082,
"step": 369
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 442.0,
"completions/mean_length": 204.84375,
"completions/mean_terminated_length": 194.9354705810547,
"completions/min_length": 65.0,
"completions/min_terminated_length": 65.0,
"epoch": 0.296,
"grad_norm": 4.286995887756348,
"kl": 0.1629638671875,
"learning_rate": 1e-06,
"loss": -0.0786,
"num_tokens": 4701822.0,
"reward": 9.278564453125,
"reward_std": 7.017640590667725,
"rewards/rm_reward_func/mean": 9.278564453125,
"rewards/rm_reward_func/std": 11.16659164428711,
"step": 370
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 396.0,
"completions/mean_length": 137.71875,
"completions/mean_terminated_length": 125.64515686035156,
"completions/min_length": 41.0,
"completions/min_terminated_length": 41.0,
"epoch": 0.2968,
"grad_norm": 6.7198004722595215,
"kl": 0.17926025390625,
"learning_rate": 1e-06,
"loss": 0.0636,
"num_tokens": 4709437.0,
"reward": 1.2930221557617188,
"reward_std": 2.7805871963500977,
"rewards/rm_reward_func/mean": 1.2930221557617188,
"rewards/rm_reward_func/std": 6.2812652587890625,
"step": 371
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 127.0,
"completions/max_terminated_length": 127.0,
"completions/mean_length": 67.65625,
"completions/mean_terminated_length": 67.65625,
"completions/min_length": 9.0,
"completions/min_terminated_length": 9.0,
"epoch": 0.2976,
"grad_norm": 7.943471908569336,
"kl": 0.2227783203125,
"learning_rate": 1e-06,
"loss": -0.0281,
"num_tokens": 4716018.0,
"reward": 0.7178955078125,
"reward_std": 2.993560314178467,
"rewards/rm_reward_func/mean": 0.7178955078125,
"rewards/rm_reward_func/std": 7.339946746826172,
"step": 372
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 501.0,
"completions/mean_length": 326.53125,
"completions/mean_terminated_length": 314.16668701171875,
"completions/min_length": 68.0,
"completions/min_terminated_length": 68.0,
"epoch": 0.2984,
"grad_norm": 2.898266315460205,
"kl": 0.1676025390625,
"learning_rate": 1e-06,
"loss": 0.0155,
"num_tokens": 4729731.0,
"reward": 8.0377197265625,
"reward_std": 4.975411891937256,
"rewards/rm_reward_func/mean": 8.0377197265625,
"rewards/rm_reward_func/std": 11.102714538574219,
"step": 373
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 504.0,
"completions/mean_length": 314.59375,
"completions/mean_terminated_length": 224.8636474609375,
"completions/min_length": 42.0,
"completions/min_terminated_length": 42.0,
"epoch": 0.2992,
"grad_norm": 5.358304023742676,
"kl": 0.1610107421875,
"learning_rate": 1e-06,
"loss": 0.197,
"num_tokens": 4742030.0,
"reward": 5.0345458984375,
"reward_std": 6.491810321807861,
"rewards/rm_reward_func/mean": 5.0345458984375,
"rewards/rm_reward_func/std": 8.370210647583008,
"step": 374
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.46875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 426.375,
"completions/mean_terminated_length": 350.8235168457031,
"completions/min_length": 92.0,
"completions/min_terminated_length": 92.0,
"epoch": 0.3,
"grad_norm": 4.6803483963012695,
"kl": 0.0802001953125,
"learning_rate": 1e-06,
"loss": -0.0817,
"num_tokens": 4761410.0,
"reward": 10.879684448242188,
"reward_std": 8.08679485321045,
"rewards/rm_reward_func/mean": 10.879684448242188,
"rewards/rm_reward_func/std": 10.522982597351074,
"step": 375
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 342.0,
"completions/mean_length": 204.09375,
"completions/mean_terminated_length": 194.16128540039062,
"completions/min_length": 50.0,
"completions/min_terminated_length": 50.0,
"epoch": 0.3008,
"grad_norm": 3.7969260215759277,
"kl": 0.1573486328125,
"learning_rate": 1e-06,
"loss": -0.011,
"num_tokens": 4770525.0,
"reward": 9.822998046875,
"reward_std": 6.23846435546875,
"rewards/rm_reward_func/mean": 9.822998046875,
"rewards/rm_reward_func/std": 9.928614616394043,
"step": 376
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 496.0,
"completions/mean_length": 231.34375,
"completions/mean_terminated_length": 212.6333465576172,
"completions/min_length": 70.0,
"completions/min_terminated_length": 70.0,
"epoch": 0.3016,
"grad_norm": 4.173020839691162,
"kl": 0.240234375,
"learning_rate": 1e-06,
"loss": -0.0314,
"num_tokens": 4780584.0,
"reward": 7.1978607177734375,
"reward_std": 5.048998832702637,
"rewards/rm_reward_func/mean": 7.1978607177734375,
"rewards/rm_reward_func/std": 9.145500183105469,
"step": 377
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 496.0,
"completions/mean_length": 312.96875,
"completions/mean_terminated_length": 246.625,
"completions/min_length": 76.0,
"completions/min_terminated_length": 76.0,
"epoch": 0.3024,
"grad_norm": 3.5591626167297363,
"kl": 0.175537109375,
"learning_rate": 1e-06,
"loss": 0.0296,
"num_tokens": 4792983.0,
"reward": 1.46685791015625,
"reward_std": 3.8416523933410645,
"rewards/rm_reward_func/mean": 1.46685791015625,
"rewards/rm_reward_func/std": 8.133792877197266,
"step": 378
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 497.0,
"completions/max_terminated_length": 497.0,
"completions/mean_length": 240.4375,
"completions/mean_terminated_length": 240.4375,
"completions/min_length": 51.0,
"completions/min_terminated_length": 51.0,
"epoch": 0.3032,
"grad_norm": 4.513542652130127,
"kl": 0.148193359375,
"learning_rate": 1e-06,
"loss": -0.0096,
"num_tokens": 4803181.0,
"reward": 3.9730072021484375,
"reward_std": 3.2463040351867676,
"rewards/rm_reward_func/mean": 3.9730072021484375,
"rewards/rm_reward_func/std": 7.680938243865967,
"step": 379
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 274.78125,
"completions/mean_terminated_length": 250.2413787841797,
"completions/min_length": 59.0,
"completions/min_terminated_length": 59.0,
"epoch": 0.304,
"grad_norm": 4.1872639656066895,
"kl": 0.088134765625,
"learning_rate": 1e-06,
"loss": 0.2139,
"num_tokens": 4816062.0,
"reward": 8.164306640625,
"reward_std": 9.427093505859375,
"rewards/rm_reward_func/mean": 8.164306640625,
"rewards/rm_reward_func/std": 11.66413402557373,
"step": 380
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 318.28125,
"completions/mean_terminated_length": 242.478271484375,
"completions/min_length": 62.0,
"completions/min_terminated_length": 62.0,
"epoch": 0.3048,
"grad_norm": 4.635010242462158,
"kl": 0.1771240234375,
"learning_rate": 1e-06,
"loss": -0.0366,
"num_tokens": 4828711.0,
"reward": 7.5330810546875,
"reward_std": 4.666987419128418,
"rewards/rm_reward_func/mean": 7.5330810546875,
"rewards/rm_reward_func/std": 6.908311367034912,
"step": 381
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 505.0,
"completions/mean_length": 274.90625,
"completions/mean_terminated_length": 220.19232177734375,
"completions/min_length": 63.0,
"completions/min_terminated_length": 63.0,
"epoch": 0.3056,
"grad_norm": 4.149857044219971,
"kl": 0.168212890625,
"learning_rate": 1e-06,
"loss": 0.0295,
"num_tokens": 4839428.0,
"reward": -3.975341796875,
"reward_std": 4.726168632507324,
"rewards/rm_reward_func/mean": -3.975341796875,
"rewards/rm_reward_func/std": 9.026304244995117,
"step": 382
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 502.0,
"completions/mean_length": 276.0,
"completions/mean_terminated_length": 209.9199981689453,
"completions/min_length": 65.0,
"completions/min_terminated_length": 65.0,
"epoch": 0.3064,
"grad_norm": 3.875955820083618,
"kl": 0.13720703125,
"learning_rate": 1e-06,
"loss": 0.0275,
"num_tokens": 4850612.0,
"reward": -5.6435546875,
"reward_std": 3.571234941482544,
"rewards/rm_reward_func/mean": -5.6435546875,
"rewards/rm_reward_func/std": 16.429338455200195,
"step": 383
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 483.0,
"completions/mean_length": 227.34375,
"completions/mean_terminated_length": 208.36668395996094,
"completions/min_length": 71.0,
"completions/min_terminated_length": 71.0,
"epoch": 0.3072,
"grad_norm": 3.766765832901001,
"kl": 0.154052734375,
"learning_rate": 1e-06,
"loss": 0.1151,
"num_tokens": 4862007.0,
"reward": 9.3204345703125,
"reward_std": 5.31584358215332,
"rewards/rm_reward_func/mean": 9.3204345703125,
"rewards/rm_reward_func/std": 12.34428596496582,
"step": 384
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 274.90625,
"completions/mean_terminated_length": 220.19232177734375,
"completions/min_length": 76.0,
"completions/min_terminated_length": 76.0,
"epoch": 0.308,
"grad_norm": 3.00742506980896,
"kl": 0.134521484375,
"learning_rate": 1e-06,
"loss": -0.0583,
"num_tokens": 4873196.0,
"reward": -3.42340087890625,
"reward_std": 3.529475450515747,
"rewards/rm_reward_func/mean": -3.42340087890625,
"rewards/rm_reward_func/std": 14.037924766540527,
"step": 385
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 507.0,
"completions/mean_length": 375.53125,
"completions/mean_terminated_length": 293.6499938964844,
"completions/min_length": 76.0,
"completions/min_terminated_length": 76.0,
"epoch": 0.3088,
"grad_norm": 4.225069046020508,
"kl": 0.15325927734375,
"learning_rate": 1e-06,
"loss": -0.1237,
"num_tokens": 4889389.0,
"reward": 10.78369140625,
"reward_std": 9.285670280456543,
"rewards/rm_reward_func/mean": 10.78369140625,
"rewards/rm_reward_func/std": 19.006528854370117,
"step": 386
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 497.0,
"completions/mean_length": 340.40625,
"completions/mean_terminated_length": 262.4090881347656,
"completions/min_length": 65.0,
"completions/min_terminated_length": 65.0,
"epoch": 0.3096,
"grad_norm": 2.9847397804260254,
"kl": 0.14892578125,
"learning_rate": 1e-06,
"loss": 0.0144,
"num_tokens": 4903578.0,
"reward": 9.98577880859375,
"reward_std": 4.354106426239014,
"rewards/rm_reward_func/mean": 9.98577880859375,
"rewards/rm_reward_func/std": 9.918432235717773,
"step": 387
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 498.0,
"completions/mean_length": 282.96875,
"completions/mean_terminated_length": 259.2758483886719,
"completions/min_length": 50.0,
"completions/min_terminated_length": 50.0,
"epoch": 0.3104,
"grad_norm": 3.2981696128845215,
"kl": 0.138427734375,
"learning_rate": 1e-06,
"loss": 0.0006,
"num_tokens": 4915585.0,
"reward": 6.951171875,
"reward_std": 4.37175178527832,
"rewards/rm_reward_func/mean": 6.951171875,
"rewards/rm_reward_func/std": 16.657833099365234,
"step": 388
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 501.0,
"completions/mean_length": 282.0625,
"completions/mean_terminated_length": 258.2758483886719,
"completions/min_length": 41.0,
"completions/min_terminated_length": 41.0,
"epoch": 0.3112,
"grad_norm": 4.533021450042725,
"kl": 0.1182861328125,
"learning_rate": 1e-06,
"loss": -0.1073,
"num_tokens": 4929283.0,
"reward": 1.020263671875,
"reward_std": 5.0986127853393555,
"rewards/rm_reward_func/mean": 1.020263671875,
"rewards/rm_reward_func/std": 10.977892875671387,
"step": 389
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 511.0,
"completions/mean_length": 252.9375,
"completions/mean_terminated_length": 193.1538543701172,
"completions/min_length": 67.0,
"completions/min_terminated_length": 67.0,
"epoch": 0.312,
"grad_norm": 4.486575603485107,
"kl": 0.15936279296875,
"learning_rate": 1e-06,
"loss": -0.0062,
"num_tokens": 4941017.0,
"reward": -1.0895843505859375,
"reward_std": 5.964107990264893,
"rewards/rm_reward_func/mean": -1.0895843505859375,
"rewards/rm_reward_func/std": 11.993568420410156,
"step": 390
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 404.875,
"completions/mean_terminated_length": 380.15386962890625,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.3128,
"grad_norm": 3.2242486476898193,
"kl": 0.097900390625,
"learning_rate": 1e-06,
"loss": 0.0056,
"num_tokens": 4955989.0,
"reward": 17.81597900390625,
"reward_std": 8.450302124023438,
"rewards/rm_reward_func/mean": 17.81597900390625,
"rewards/rm_reward_func/std": 14.226698875427246,
"step": 391
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 500.0,
"completions/mean_length": 275.53125,
"completions/mean_terminated_length": 220.9615478515625,
"completions/min_length": 66.0,
"completions/min_terminated_length": 66.0,
"epoch": 0.3136,
"grad_norm": 3.7851226329803467,
"kl": 0.231781005859375,
"learning_rate": 1e-06,
"loss": 0.0319,
"num_tokens": 4967710.0,
"reward": 13.55126953125,
"reward_std": 4.914253234863281,
"rewards/rm_reward_func/mean": 13.55126953125,
"rewards/rm_reward_func/std": 15.68505859375,
"step": 392
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 479.0,
"completions/mean_length": 208.75,
"completions/mean_terminated_length": 188.53334045410156,
"completions/min_length": 71.0,
"completions/min_terminated_length": 71.0,
"epoch": 0.3144,
"grad_norm": 3.546250104904175,
"kl": 0.248779296875,
"learning_rate": 1e-06,
"loss": 0.1134,
"num_tokens": 4977470.0,
"reward": 6.796875,
"reward_std": 4.045927047729492,
"rewards/rm_reward_func/mean": 6.796875,
"rewards/rm_reward_func/std": 9.776312828063965,
"step": 393
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 285.1875,
"completions/mean_terminated_length": 252.7857208251953,
"completions/min_length": 71.0,
"completions/min_terminated_length": 71.0,
"epoch": 0.3152,
"grad_norm": 8.546547889709473,
"kl": 0.190673828125,
"learning_rate": 1e-06,
"loss": 0.2501,
"num_tokens": 4990396.0,
"reward": 12.95703125,
"reward_std": 6.034456729888916,
"rewards/rm_reward_func/mean": 12.95703125,
"rewards/rm_reward_func/std": 11.617748260498047,
"step": 394
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 449.0,
"completions/max_terminated_length": 449.0,
"completions/mean_length": 294.84375,
"completions/mean_terminated_length": 294.84375,
"completions/min_length": 125.0,
"completions/min_terminated_length": 125.0,
"epoch": 0.316,
"grad_norm": 3.5022923946380615,
"kl": 0.162109375,
"learning_rate": 1e-06,
"loss": -0.0858,
"num_tokens": 5002815.0,
"reward": 14.046142578125,
"reward_std": 6.6521525382995605,
"rewards/rm_reward_func/mean": 14.046142578125,
"rewards/rm_reward_func/std": 15.969138145446777,
"step": 395
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.40625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 492.0,
"completions/mean_length": 406.75,
"completions/mean_terminated_length": 334.7368469238281,
"completions/min_length": 179.0,
"completions/min_terminated_length": 179.0,
"epoch": 0.3168,
"grad_norm": 3.3227157592773438,
"kl": 0.09228515625,
"learning_rate": 1e-06,
"loss": -0.0137,
"num_tokens": 5021671.0,
"reward": 3.28466796875,
"reward_std": 4.573013782501221,
"rewards/rm_reward_func/mean": 3.28466796875,
"rewards/rm_reward_func/std": 10.546538352966309,
"step": 396
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 480.0,
"completions/max_terminated_length": 480.0,
"completions/mean_length": 199.3125,
"completions/mean_terminated_length": 199.3125,
"completions/min_length": 57.0,
"completions/min_terminated_length": 57.0,
"epoch": 0.3176,
"grad_norm": 8.376974105834961,
"kl": 0.1419677734375,
"learning_rate": 1e-06,
"loss": 0.01,
"num_tokens": 5033201.0,
"reward": 3.65576171875,
"reward_std": 4.244373321533203,
"rewards/rm_reward_func/mean": 3.65576171875,
"rewards/rm_reward_func/std": 15.917009353637695,
"step": 397
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 478.0,
"completions/max_terminated_length": 478.0,
"completions/mean_length": 303.9375,
"completions/mean_terminated_length": 303.9375,
"completions/min_length": 110.0,
"completions/min_terminated_length": 110.0,
"epoch": 0.3184,
"grad_norm": 3.5459303855895996,
"kl": 0.10992431640625,
"learning_rate": 1e-06,
"loss": 0.0089,
"num_tokens": 5044935.0,
"reward": 5.091552734375,
"reward_std": 4.916255950927734,
"rewards/rm_reward_func/mean": 5.091552734375,
"rewards/rm_reward_func/std": 14.6499605178833,
"step": 398
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 317.90625,
"completions/mean_terminated_length": 281.96295166015625,
"completions/min_length": 72.0,
"completions/min_terminated_length": 72.0,
"epoch": 0.3192,
"grad_norm": 3.4768733978271484,
"kl": 0.1575927734375,
"learning_rate": 1e-06,
"loss": 0.0603,
"num_tokens": 5058220.0,
"reward": 2.6343994140625,
"reward_std": 6.248598098754883,
"rewards/rm_reward_func/mean": 2.6343994140625,
"rewards/rm_reward_func/std": 12.041556358337402,
"step": 399
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 495.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 312.09375,
"completions/mean_terminated_length": 312.09375,
"completions/min_length": 106.0,
"completions/min_terminated_length": 106.0,
"epoch": 0.32,
"grad_norm": 3.6274449825286865,
"kl": 0.13006591796875,
"learning_rate": 1e-06,
"loss": 0.0541,
"num_tokens": 5070567.0,
"reward": 6.393798828125,
"reward_std": 4.916718006134033,
"rewards/rm_reward_func/mean": 6.393798828125,
"rewards/rm_reward_func/std": 14.114630699157715,
"step": 400
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 120.0,
"completions/max_terminated_length": 120.0,
"completions/mean_length": 77.9375,
"completions/mean_terminated_length": 77.9375,
"completions/min_length": 31.0,
"completions/min_terminated_length": 31.0,
"epoch": 0.3208,
"grad_norm": 4.715888023376465,
"kl": 0.33203125,
"learning_rate": 1e-06,
"loss": -0.0051,
"num_tokens": 5077253.0,
"reward": 2.888671875,
"reward_std": 3.143357038497925,
"rewards/rm_reward_func/mean": 2.888671875,
"rewards/rm_reward_func/std": 10.683290481567383,
"step": 401
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 297.0,
"completions/mean_length": 140.5625,
"completions/mean_terminated_length": 115.80000305175781,
"completions/min_length": 37.0,
"completions/min_terminated_length": 37.0,
"epoch": 0.3216,
"grad_norm": 7.256206512451172,
"kl": 0.289306640625,
"learning_rate": 1e-06,
"loss": 0.432,
"num_tokens": 5085031.0,
"reward": 2.5806884765625,
"reward_std": 4.735476016998291,
"rewards/rm_reward_func/mean": 2.5806884765625,
"rewards/rm_reward_func/std": 8.712172508239746,
"step": 402
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 465.0,
"completions/mean_length": 286.125,
"completions/mean_terminated_length": 262.75860595703125,
"completions/min_length": 75.0,
"completions/min_terminated_length": 75.0,
"epoch": 0.3224,
"grad_norm": 6.090017795562744,
"kl": 0.133056640625,
"learning_rate": 1e-06,
"loss": -0.2161,
"num_tokens": 5097227.0,
"reward": 3.853515625,
"reward_std": 5.223775863647461,
"rewards/rm_reward_func/mean": 3.853515625,
"rewards/rm_reward_func/std": 14.418701171875,
"step": 403
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 503.0,
"completions/mean_length": 303.3125,
"completions/mean_terminated_length": 281.7241516113281,
"completions/min_length": 26.0,
"completions/min_terminated_length": 26.0,
"epoch": 0.3232,
"grad_norm": 4.603776931762695,
"kl": 0.0992431640625,
"learning_rate": 1e-06,
"loss": 0.0173,
"num_tokens": 5109629.0,
"reward": 6.81072998046875,
"reward_std": 10.010828018188477,
"rewards/rm_reward_func/mean": 6.81072998046875,
"rewards/rm_reward_func/std": 19.7315673828125,
"step": 404
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 327.40625,
"completions/mean_terminated_length": 265.875,
"completions/min_length": 39.0,
"completions/min_terminated_length": 39.0,
"epoch": 0.324,
"grad_norm": 3.5028727054595947,
"kl": 0.10601806640625,
"learning_rate": 1e-06,
"loss": -0.067,
"num_tokens": 5122570.0,
"reward": -3.7301025390625,
"reward_std": 4.875129222869873,
"rewards/rm_reward_func/mean": -3.7301025390625,
"rewards/rm_reward_func/std": 7.149316787719727,
"step": 405
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 360.21875,
"completions/mean_terminated_length": 280.71429443359375,
"completions/min_length": 66.0,
"completions/min_terminated_length": 66.0,
"epoch": 0.3248,
"grad_norm": 5.7773871421813965,
"kl": 0.1090087890625,
"learning_rate": 1e-06,
"loss": -0.3303,
"num_tokens": 5136473.0,
"reward": 0.67950439453125,
"reward_std": 9.863033294677734,
"rewards/rm_reward_func/mean": 0.67950439453125,
"rewards/rm_reward_func/std": 14.524174690246582,
"step": 406
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 461.0,
"completions/mean_length": 238.84375,
"completions/mean_terminated_length": 162.36000061035156,
"completions/min_length": 70.0,
"completions/min_terminated_length": 70.0,
"epoch": 0.3256,
"grad_norm": 2.8504834175109863,
"kl": 0.235107421875,
"learning_rate": 1e-06,
"loss": -0.0181,
"num_tokens": 5147908.0,
"reward": 5.07373046875,
"reward_std": 3.009080648422241,
"rewards/rm_reward_func/mean": 5.07373046875,
"rewards/rm_reward_func/std": 7.276069641113281,
"step": 407
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 490.0,
"completions/mean_length": 342.375,
"completions/mean_terminated_length": 285.8333435058594,
"completions/min_length": 72.0,
"completions/min_terminated_length": 72.0,
"epoch": 0.3264,
"grad_norm": 6.275087833404541,
"kl": 0.151123046875,
"learning_rate": 1e-06,
"loss": 0.237,
"num_tokens": 5161120.0,
"reward": 6.7158203125,
"reward_std": 6.726596832275391,
"rewards/rm_reward_func/mean": 6.7158203125,
"rewards/rm_reward_func/std": 7.3756632804870605,
"step": 408
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 417.15625,
"completions/mean_terminated_length": 390.6000061035156,
"completions/min_length": 235.0,
"completions/min_terminated_length": 235.0,
"epoch": 0.3272,
"grad_norm": 3.3832011222839355,
"kl": 0.0989990234375,
"learning_rate": 1e-06,
"loss": 0.0316,
"num_tokens": 5178797.0,
"reward": 6.41094970703125,
"reward_std": 5.119528293609619,
"rewards/rm_reward_func/mean": 6.41094970703125,
"rewards/rm_reward_func/std": 7.976118564605713,
"step": 409
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 425.0,
"completions/mean_length": 293.25,
"completions/mean_terminated_length": 286.19354248046875,
"completions/min_length": 76.0,
"completions/min_terminated_length": 76.0,
"epoch": 0.328,
"grad_norm": 3.2517099380493164,
"kl": 0.17236328125,
"learning_rate": 1e-06,
"loss": 0.0392,
"num_tokens": 5190149.0,
"reward": 9.473388671875,
"reward_std": 6.672955513000488,
"rewards/rm_reward_func/mean": 9.473388671875,
"rewards/rm_reward_func/std": 11.397151947021484,
"step": 410
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 407.0,
"completions/mean_length": 304.125,
"completions/mean_terminated_length": 209.63636779785156,
"completions/min_length": 71.0,
"completions/min_terminated_length": 71.0,
"epoch": 0.3288,
"grad_norm": 6.965356349945068,
"kl": 0.15704345703125,
"learning_rate": 1e-06,
"loss": 0.2693,
"num_tokens": 5203601.0,
"reward": -2.846466064453125,
"reward_std": 3.277106761932373,
"rewards/rm_reward_func/mean": -2.846466064453125,
"rewards/rm_reward_func/std": 8.159171104431152,
"step": 411
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 452.0,
"completions/mean_length": 329.3125,
"completions/mean_terminated_length": 268.41668701171875,
"completions/min_length": 44.0,
"completions/min_terminated_length": 44.0,
"epoch": 0.3296,
"grad_norm": 3.6799185276031494,
"kl": 0.079345703125,
"learning_rate": 1e-06,
"loss": -0.0095,
"num_tokens": 5217035.0,
"reward": -4.873779296875,
"reward_std": 5.421168327331543,
"rewards/rm_reward_func/mean": -4.873779296875,
"rewards/rm_reward_func/std": 10.083694458007812,
"step": 412
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 466.0,
"completions/mean_length": 382.3125,
"completions/mean_terminated_length": 358.2962951660156,
"completions/min_length": 149.0,
"completions/min_terminated_length": 149.0,
"epoch": 0.3304,
"grad_norm": 3.278068780899048,
"kl": 0.1441650390625,
"learning_rate": 1e-06,
"loss": 0.0341,
"num_tokens": 5232501.0,
"reward": 10.257858276367188,
"reward_std": 8.294170379638672,
"rewards/rm_reward_func/mean": 10.257858276367188,
"rewards/rm_reward_func/std": 13.173551559448242,
"step": 413
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 482.0,
"completions/mean_length": 304.84375,
"completions/mean_terminated_length": 235.7916717529297,
"completions/min_length": 54.0,
"completions/min_terminated_length": 54.0,
"epoch": 0.3312,
"grad_norm": 3.0338048934936523,
"kl": 0.15618896484375,
"learning_rate": 1e-06,
"loss": -0.1137,
"num_tokens": 5244808.0,
"reward": 5.3134765625,
"reward_std": 6.2857160568237305,
"rewards/rm_reward_func/mean": 5.3134765625,
"rewards/rm_reward_func/std": 16.0675106048584,
"step": 414
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 504.0,
"completions/mean_length": 249.125,
"completions/mean_terminated_length": 200.44444274902344,
"completions/min_length": 67.0,
"completions/min_terminated_length": 67.0,
"epoch": 0.332,
"grad_norm": 6.477999687194824,
"kl": 0.2847900390625,
"learning_rate": 1e-06,
"loss": 0.1968,
"num_tokens": 5256652.0,
"reward": 9.817138671875,
"reward_std": 10.444900512695312,
"rewards/rm_reward_func/mean": 9.817138671875,
"rewards/rm_reward_func/std": 11.969996452331543,
"step": 415
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 463.0,
"completions/mean_length": 409.21875,
"completions/mean_terminated_length": 347.5500183105469,
"completions/min_length": 217.0,
"completions/min_terminated_length": 217.0,
"epoch": 0.3328,
"grad_norm": 3.6648788452148438,
"kl": 0.2205810546875,
"learning_rate": 1e-06,
"loss": -0.0097,
"num_tokens": 5273531.0,
"reward": -2.630218505859375,
"reward_std": 3.172177314758301,
"rewards/rm_reward_func/mean": -2.630218505859375,
"rewards/rm_reward_func/std": 7.456665515899658,
"step": 416
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 338.96875,
"completions/mean_terminated_length": 314.25,
"completions/min_length": 68.0,
"completions/min_terminated_length": 68.0,
"epoch": 0.3336,
"grad_norm": 4.057036876678467,
"kl": 0.1800537109375,
"learning_rate": 1e-06,
"loss": 0.0165,
"num_tokens": 5288122.0,
"reward": 7.453369140625,
"reward_std": 4.4758758544921875,
"rewards/rm_reward_func/mean": 7.453369140625,
"rewards/rm_reward_func/std": 8.864148139953613,
"step": 417
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 487.0,
"completions/mean_length": 271.8125,
"completions/mean_terminated_length": 264.06451416015625,
"completions/min_length": 55.0,
"completions/min_terminated_length": 55.0,
"epoch": 0.3344,
"grad_norm": 7.1819305419921875,
"kl": 0.20086669921875,
"learning_rate": 1e-06,
"loss": 0.258,
"num_tokens": 5302668.0,
"reward": 4.56103515625,
"reward_std": 5.861100673675537,
"rewards/rm_reward_func/mean": 4.56103515625,
"rewards/rm_reward_func/std": 10.66663932800293,
"step": 418
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 502.0,
"completions/mean_length": 449.9375,
"completions/mean_terminated_length": 421.727294921875,
"completions/min_length": 332.0,
"completions/min_terminated_length": 332.0,
"epoch": 0.3352,
"grad_norm": 4.249541282653809,
"kl": 0.2069091796875,
"learning_rate": 1e-06,
"loss": 0.0362,
"num_tokens": 5319242.0,
"reward": 11.88818359375,
"reward_std": 10.299996376037598,
"rewards/rm_reward_func/mean": 11.88818359375,
"rewards/rm_reward_func/std": 14.682938575744629,
"step": 419
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 301.375,
"completions/mean_terminated_length": 294.58062744140625,
"completions/min_length": 107.0,
"completions/min_terminated_length": 107.0,
"epoch": 0.336,
"grad_norm": 4.146368980407715,
"kl": 0.2005615234375,
"learning_rate": 1e-06,
"loss": 0.1198,
"num_tokens": 5334606.0,
"reward": 1.5439453125,
"reward_std": 5.175732612609863,
"rewards/rm_reward_func/mean": 1.5439453125,
"rewards/rm_reward_func/std": 11.174506187438965,
"step": 420
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 491.0,
"completions/mean_length": 240.125,
"completions/mean_terminated_length": 222.00001525878906,
"completions/min_length": 58.0,
"completions/min_terminated_length": 58.0,
"epoch": 0.3368,
"grad_norm": 5.178651332855225,
"kl": 0.252685546875,
"learning_rate": 1e-06,
"loss": 0.3182,
"num_tokens": 5345146.0,
"reward": -5.8485107421875,
"reward_std": 10.920194625854492,
"rewards/rm_reward_func/mean": -5.8485107421875,
"rewards/rm_reward_func/std": 11.926911354064941,
"step": 421
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 258.625,
"completions/mean_terminated_length": 222.4285888671875,
"completions/min_length": 68.0,
"completions/min_terminated_length": 68.0,
"epoch": 0.3376,
"grad_norm": 7.027132511138916,
"kl": 0.424560546875,
"learning_rate": 1e-06,
"loss": 0.2645,
"num_tokens": 5356774.0,
"reward": 10.3265380859375,
"reward_std": 7.619661331176758,
"rewards/rm_reward_func/mean": 10.3265380859375,
"rewards/rm_reward_func/std": 16.47081184387207,
"step": 422
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 501.0,
"completions/mean_length": 366.71875,
"completions/mean_terminated_length": 333.19232177734375,
"completions/min_length": 184.0,
"completions/min_terminated_length": 184.0,
"epoch": 0.3384,
"grad_norm": 5.974679470062256,
"kl": 0.1844482421875,
"learning_rate": 1e-06,
"loss": 0.0469,
"num_tokens": 5370701.0,
"reward": 20.43310546875,
"reward_std": 9.503026008605957,
"rewards/rm_reward_func/mean": 20.43310546875,
"rewards/rm_reward_func/std": 18.206256866455078,
"step": 423
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 501.0,
"completions/mean_length": 346.90625,
"completions/mean_terminated_length": 300.67999267578125,
"completions/min_length": 123.0,
"completions/min_terminated_length": 123.0,
"epoch": 0.3392,
"grad_norm": 4.590559482574463,
"kl": 0.110626220703125,
"learning_rate": 1e-06,
"loss": 0.0688,
"num_tokens": 5384434.0,
"reward": 1.7897424697875977,
"reward_std": 6.459235191345215,
"rewards/rm_reward_func/mean": 1.7897424697875977,
"rewards/rm_reward_func/std": 15.98327350616455,
"step": 424
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.4375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 496.0,
"completions/mean_length": 423.15625,
"completions/mean_terminated_length": 354.0555725097656,
"completions/min_length": 82.0,
"completions/min_terminated_length": 82.0,
"epoch": 0.34,
"grad_norm": 16.818695068359375,
"kl": 0.2874755859375,
"learning_rate": 1e-06,
"loss": -0.0521,
"num_tokens": 5400303.0,
"reward": 3.7025909423828125,
"reward_std": 7.56202507019043,
"rewards/rm_reward_func/mean": 3.7025909423828125,
"rewards/rm_reward_func/std": 20.20536231994629,
"step": 425
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 353.71875,
"completions/mean_terminated_length": 281.7727355957031,
"completions/min_length": 63.0,
"completions/min_terminated_length": 63.0,
"epoch": 0.3408,
"grad_norm": 5.755769729614258,
"kl": 0.47705078125,
"learning_rate": 1e-06,
"loss": 0.2143,
"num_tokens": 5413798.0,
"reward": -4.92926025390625,
"reward_std": 6.647279739379883,
"rewards/rm_reward_func/mean": -4.92926025390625,
"rewards/rm_reward_func/std": 9.619757652282715,
"step": 426
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.40625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 436.0,
"completions/mean_length": 363.4375,
"completions/mean_terminated_length": 261.78948974609375,
"completions/min_length": 138.0,
"completions/min_terminated_length": 138.0,
"epoch": 0.3416,
"grad_norm": 5.360143661499023,
"kl": 0.496826171875,
"learning_rate": 1e-06,
"loss": 0.1156,
"num_tokens": 5427236.0,
"reward": -9.527587890625,
"reward_std": 7.441409111022949,
"rewards/rm_reward_func/mean": -9.527587890625,
"rewards/rm_reward_func/std": 8.035903930664062,
"step": 427
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 399.0,
"completions/mean_length": 301.53125,
"completions/mean_terminated_length": 191.2857208251953,
"completions/min_length": 23.0,
"completions/min_terminated_length": 23.0,
"epoch": 0.3424,
"grad_norm": 5.262341022491455,
"kl": 0.5040283203125,
"learning_rate": 1e-06,
"loss": 0.1718,
"num_tokens": 5441501.0,
"reward": -8.2998046875,
"reward_std": 4.3596343994140625,
"rewards/rm_reward_func/mean": -8.2998046875,
"rewards/rm_reward_func/std": 10.656414031982422,
"step": 428
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.75,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 498.40625,
"completions/mean_terminated_length": 457.625,
"completions/min_length": 409.0,
"completions/min_terminated_length": 409.0,
"epoch": 0.3432,
"grad_norm": 5.260901927947998,
"kl": 0.877197265625,
"learning_rate": 1e-06,
"loss": 0.0611,
"num_tokens": 5460434.0,
"reward": -9.208251953125,
"reward_std": 7.153204917907715,
"rewards/rm_reward_func/mean": -9.208251953125,
"rewards/rm_reward_func/std": 9.89172077178955,
"step": 429
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 497.0,
"completions/mean_length": 246.0625,
"completions/mean_terminated_length": 208.07144165039062,
"completions/min_length": 22.0,
"completions/min_terminated_length": 22.0,
"epoch": 0.344,
"grad_norm": 7.583652496337891,
"kl": 0.421142578125,
"learning_rate": 1e-06,
"loss": 0.2763,
"num_tokens": 5470404.0,
"reward": -1.38763427734375,
"reward_std": 8.258423805236816,
"rewards/rm_reward_func/mean": -1.38763427734375,
"rewards/rm_reward_func/std": 14.190618515014648,
"step": 430
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.59375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 499.0,
"completions/mean_length": 375.84375,
"completions/mean_terminated_length": 176.84616088867188,
"completions/min_length": 61.0,
"completions/min_terminated_length": 61.0,
"epoch": 0.3448,
"grad_norm": 7.52949857711792,
"kl": 1.322265625,
"learning_rate": 1e-06,
"loss": 0.3466,
"num_tokens": 5486727.0,
"reward": -10.94189453125,
"reward_std": 8.947233200073242,
"rewards/rm_reward_func/mean": -10.94189453125,
"rewards/rm_reward_func/std": 11.468242645263672,
"step": 431
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.5,
"completions/max_length": 512.0,
"completions/max_terminated_length": 373.0,
"completions/mean_length": 336.09375,
"completions/mean_terminated_length": 160.1875,
"completions/min_length": 18.0,
"completions/min_terminated_length": 18.0,
"epoch": 0.3456,
"grad_norm": 6.60636568069458,
"kl": 1.8037109375,
"learning_rate": 1e-06,
"loss": 0.1765,
"num_tokens": 5503626.0,
"reward": -11.5931396484375,
"reward_std": 7.840054035186768,
"rewards/rm_reward_func/mean": -11.5931396484375,
"rewards/rm_reward_func/std": 13.185185432434082,
"step": 432
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 371.0,
"completions/mean_length": 373.71875,
"completions/mean_terminated_length": 143.25,
"completions/min_length": 56.0,
"completions/min_terminated_length": 56.0,
"epoch": 0.3464,
"grad_norm": 20.934152603149414,
"kl": 1.482421875,
"learning_rate": 1e-06,
"loss": 0.4161,
"num_tokens": 5518233.0,
"reward": -11.6781005859375,
"reward_std": 9.535715103149414,
"rewards/rm_reward_func/mean": -11.6781005859375,
"rewards/rm_reward_func/std": 12.062625885009766,
"step": 433
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 415.0,
"completions/mean_length": 397.625,
"completions/mean_terminated_length": 207.0,
"completions/min_length": 76.0,
"completions/min_terminated_length": 76.0,
"epoch": 0.3472,
"grad_norm": 6.227436542510986,
"kl": 1.304443359375,
"learning_rate": 1e-06,
"loss": 0.1463,
"num_tokens": 5535941.0,
"reward": -11.416015625,
"reward_std": 10.354272842407227,
"rewards/rm_reward_func/mean": -11.416015625,
"rewards/rm_reward_func/std": 10.621146202087402,
"step": 434
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.8125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 434.0,
"completions/mean_length": 450.34375,
"completions/mean_terminated_length": 183.1666717529297,
"completions/min_length": 51.0,
"completions/min_terminated_length": 51.0,
"epoch": 0.348,
"grad_norm": 24.91790199279785,
"kl": 2.39453125,
"learning_rate": 1e-06,
"loss": 0.1695,
"num_tokens": 5553336.0,
"reward": -16.6943359375,
"reward_std": 5.182467460632324,
"rewards/rm_reward_func/mean": -16.6943359375,
"rewards/rm_reward_func/std": 7.9551005363464355,
"step": 435
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.84375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 496.6875,
"completions/mean_terminated_length": 414.0,
"completions/min_length": 306.0,
"completions/min_terminated_length": 306.0,
"epoch": 0.3488,
"grad_norm": 7.014784812927246,
"kl": 2.12548828125,
"learning_rate": 1e-06,
"loss": 0.1083,
"num_tokens": 5571502.0,
"reward": -18.14990234375,
"reward_std": 7.876721382141113,
"rewards/rm_reward_func/mean": -18.14990234375,
"rewards/rm_reward_func/std": 9.936715126037598,
"step": 436
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.78125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 453.0,
"completions/mean_length": 435.375,
"completions/mean_terminated_length": 161.71429443359375,
"completions/min_length": 74.0,
"completions/min_terminated_length": 74.0,
"epoch": 0.3496,
"grad_norm": 6.28053617477417,
"kl": 3.0546875,
"learning_rate": 1e-06,
"loss": 0.3596,
"num_tokens": 5588674.0,
"reward": -16.22265625,
"reward_std": 9.346332550048828,
"rewards/rm_reward_func/mean": -16.22265625,
"rewards/rm_reward_func/std": 12.737253189086914,
"step": 437
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.84375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 431.0,
"completions/mean_length": 487.4375,
"completions/mean_terminated_length": 354.8000183105469,
"completions/min_length": 267.0,
"completions/min_terminated_length": 267.0,
"epoch": 0.3504,
"grad_norm": 6.686891078948975,
"kl": 2.892578125,
"learning_rate": 1e-06,
"loss": 0.1816,
"num_tokens": 5608344.0,
"reward": -13.66796875,
"reward_std": 10.213824272155762,
"rewards/rm_reward_func/mean": -13.66796875,
"rewards/rm_reward_func/std": 13.559255599975586,
"step": 438
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.65625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 475.03125,
"completions/mean_terminated_length": 404.4545593261719,
"completions/min_length": 330.0,
"completions/min_terminated_length": 330.0,
"epoch": 0.3512,
"grad_norm": 4.943912506103516,
"kl": 1.7698974609375,
"learning_rate": 1e-06,
"loss": 0.091,
"num_tokens": 5629521.0,
"reward": -12.60577392578125,
"reward_std": 10.764892578125,
"rewards/rm_reward_func/mean": -12.60577392578125,
"rewards/rm_reward_func/std": 13.864947319030762,
"step": 439
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.6875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 449.28125,
"completions/mean_terminated_length": 311.3000183105469,
"completions/min_length": 66.0,
"completions/min_terminated_length": 66.0,
"epoch": 0.352,
"grad_norm": 10.61517333984375,
"kl": 3.8173828125,
"learning_rate": 1e-06,
"loss": 0.1559,
"num_tokens": 5647154.0,
"reward": -16.4229736328125,
"reward_std": 4.779942989349365,
"rewards/rm_reward_func/mean": -16.4229736328125,
"rewards/rm_reward_func/std": 6.662553787231445,
"step": 440
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.65625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 459.0,
"completions/mean_length": 438.65625,
"completions/mean_terminated_length": 298.6363830566406,
"completions/min_length": 76.0,
"completions/min_terminated_length": 76.0,
"epoch": 0.3528,
"grad_norm": 9.699241638183594,
"kl": 3.1298828125,
"learning_rate": 1e-06,
"loss": 0.2842,
"num_tokens": 5664383.0,
"reward": -12.2291259765625,
"reward_std": 8.05883502960205,
"rewards/rm_reward_func/mean": -12.2291259765625,
"rewards/rm_reward_func/std": 13.489725112915039,
"step": 441
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 488.0,
"completions/mean_length": 356.9375,
"completions/mean_terminated_length": 98.5,
"completions/min_length": 46.0,
"completions/min_terminated_length": 46.0,
"epoch": 0.3536,
"grad_norm": 8.036911010742188,
"kl": 3.30029296875,
"learning_rate": 1e-06,
"loss": 0.2722,
"num_tokens": 5678285.0,
"reward": -16.7509765625,
"reward_std": 7.531893730163574,
"rewards/rm_reward_func/mean": -16.7509765625,
"rewards/rm_reward_func/std": 9.410063743591309,
"step": 442
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 484.0,
"completions/mean_length": 502.78125,
"completions/mean_terminated_length": 438.25,
"completions/min_length": 400.0,
"completions/min_terminated_length": 400.0,
"epoch": 0.3544,
"grad_norm": 11.907139778137207,
"kl": 4.3046875,
"learning_rate": 1e-06,
"loss": 0.1862,
"num_tokens": 5696630.0,
"reward": -15.4208984375,
"reward_std": 7.1098833084106445,
"rewards/rm_reward_func/mean": -15.4208984375,
"rewards/rm_reward_func/std": 17.015018463134766,
"step": 443
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.71875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 507.0,
"completions/mean_length": 423.71875,
"completions/mean_terminated_length": 198.11111450195312,
"completions/min_length": 44.0,
"completions/min_terminated_length": 44.0,
"epoch": 0.3552,
"grad_norm": 131.54908752441406,
"kl": 3.455078125,
"learning_rate": 1e-06,
"loss": 0.1955,
"num_tokens": 5713213.0,
"reward": -18.561279296875,
"reward_std": 7.316192626953125,
"rewards/rm_reward_func/mean": -18.561279296875,
"rewards/rm_reward_func/std": 9.157906532287598,
"step": 444
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.78125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 465.0,
"completions/mean_length": 462.34375,
"completions/mean_terminated_length": 285.0,
"completions/min_length": 135.0,
"completions/min_terminated_length": 135.0,
"epoch": 0.356,
"grad_norm": 5.364108085632324,
"kl": 3.9921875,
"learning_rate": 1e-06,
"loss": 0.2971,
"num_tokens": 5729992.0,
"reward": -8.744140625,
"reward_std": 15.03476333618164,
"rewards/rm_reward_func/mean": -8.744140625,
"rewards/rm_reward_func/std": 19.18375587463379,
"step": 445
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.75,
"completions/max_length": 512.0,
"completions/max_terminated_length": 476.0,
"completions/mean_length": 434.6875,
"completions/mean_terminated_length": 202.75,
"completions/min_length": 17.0,
"completions/min_terminated_length": 17.0,
"epoch": 0.3568,
"grad_norm": 8.373677253723145,
"kl": 4.99609375,
"learning_rate": 1e-06,
"loss": 0.2947,
"num_tokens": 5747238.0,
"reward": -14.09228515625,
"reward_std": 10.04998779296875,
"rewards/rm_reward_func/mean": -14.09228515625,
"rewards/rm_reward_func/std": 12.171612739562988,
"step": 446
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.59375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 463.0,
"completions/mean_length": 425.25,
"completions/mean_terminated_length": 298.4615478515625,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.3576,
"grad_norm": 6.68599271774292,
"kl": 3.321533203125,
"learning_rate": 1e-06,
"loss": 0.3343,
"num_tokens": 5764894.0,
"reward": -3.4609375,
"reward_std": 9.75390625,
"rewards/rm_reward_func/mean": -3.4609375,
"rewards/rm_reward_func/std": 23.642847061157227,
"step": 447
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 430.0,
"completions/mean_length": 314.0625,
"completions/mean_terminated_length": 210.38095092773438,
"completions/min_length": 75.0,
"completions/min_terminated_length": 75.0,
"epoch": 0.3584,
"grad_norm": 16.456398010253906,
"kl": 2.2950439453125,
"learning_rate": 1e-06,
"loss": 0.3137,
"num_tokens": 5777040.0,
"reward": -8.64706802368164,
"reward_std": 6.216998100280762,
"rewards/rm_reward_func/mean": -8.64706802368164,
"rewards/rm_reward_func/std": 15.454483032226562,
"step": 448
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.40625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 483.0,
"completions/mean_length": 356.21875,
"completions/mean_terminated_length": 249.63157653808594,
"completions/min_length": 49.0,
"completions/min_terminated_length": 49.0,
"epoch": 0.3592,
"grad_norm": 9.984318733215332,
"kl": 0.9708251953125,
"learning_rate": 1e-06,
"loss": 0.1167,
"num_tokens": 5791887.0,
"reward": 2.573716163635254,
"reward_std": 8.430018424987793,
"rewards/rm_reward_func/mean": 2.573716163635254,
"rewards/rm_reward_func/std": 14.201119422912598,
"step": 449
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 406.09375,
"completions/mean_terminated_length": 370.79168701171875,
"completions/min_length": 110.0,
"completions/min_terminated_length": 110.0,
"epoch": 0.36,
"grad_norm": 7.163238525390625,
"kl": 0.5322265625,
"learning_rate": 1e-06,
"loss": 0.0106,
"num_tokens": 5807138.0,
"reward": -2.3760223388671875,
"reward_std": 7.770718574523926,
"rewards/rm_reward_func/mean": -2.3760223388671875,
"rewards/rm_reward_func/std": 10.865222930908203,
"step": 450
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.46875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 373.0,
"completions/mean_length": 329.5,
"completions/mean_terminated_length": 168.47059631347656,
"completions/min_length": 21.0,
"completions/min_terminated_length": 21.0,
"epoch": 0.3608,
"grad_norm": 17.861509323120117,
"kl": 5.4453125,
"learning_rate": 1e-06,
"loss": 0.5361,
"num_tokens": 5822826.0,
"reward": -9.7861328125,
"reward_std": 11.642084121704102,
"rewards/rm_reward_func/mean": -9.7861328125,
"rewards/rm_reward_func/std": 16.064666748046875,
"step": 451
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 507.0,
"completions/mean_length": 445.1875,
"completions/mean_terminated_length": 333.8333435058594,
"completions/min_length": 65.0,
"completions/min_terminated_length": 65.0,
"epoch": 0.3616,
"grad_norm": 37.739498138427734,
"kl": 4.68994140625,
"learning_rate": 1e-06,
"loss": 0.2919,
"num_tokens": 5839992.0,
"reward": -1.2283172607421875,
"reward_std": 15.66540813446045,
"rewards/rm_reward_func/mean": -1.2283172607421875,
"rewards/rm_reward_func/std": 21.28140640258789,
"step": 452
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.59375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 496.0,
"completions/mean_length": 413.0,
"completions/mean_terminated_length": 268.3077087402344,
"completions/min_length": 56.0,
"completions/min_terminated_length": 56.0,
"epoch": 0.3624,
"grad_norm": 15.843510627746582,
"kl": 5.333740234375,
"learning_rate": 1e-06,
"loss": 0.2438,
"num_tokens": 5857648.0,
"reward": -10.1357421875,
"reward_std": 7.069613456726074,
"rewards/rm_reward_func/mean": -10.1357421875,
"rewards/rm_reward_func/std": 11.879778861999512,
"step": 453
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.84375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 430.0,
"completions/mean_length": 492.78125,
"completions/mean_terminated_length": 389.0,
"completions/min_length": 257.0,
"completions/min_terminated_length": 257.0,
"epoch": 0.3632,
"grad_norm": 7.660159587860107,
"kl": 6.46484375,
"learning_rate": 1e-06,
"loss": 0.3101,
"num_tokens": 5875665.0,
"reward": -10.20697021484375,
"reward_std": 12.935501098632812,
"rewards/rm_reward_func/mean": -10.20697021484375,
"rewards/rm_reward_func/std": 14.417232513427734,
"step": 454
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.6875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 267.0,
"completions/mean_length": 411.5,
"completions/mean_terminated_length": 190.40000915527344,
"completions/min_length": 78.0,
"completions/min_terminated_length": 78.0,
"epoch": 0.364,
"grad_norm": 12.648509979248047,
"kl": 8.0,
"learning_rate": 1e-06,
"loss": 0.625,
"num_tokens": 5891249.0,
"reward": -18.5673828125,
"reward_std": 8.711212158203125,
"rewards/rm_reward_func/mean": -18.5673828125,
"rewards/rm_reward_func/std": 9.04566478729248,
"step": 455
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.5,
"completions/max_length": 512.0,
"completions/max_terminated_length": 403.0,
"completions/mean_length": 350.40625,
"completions/mean_terminated_length": 188.8125,
"completions/min_length": 77.0,
"completions/min_terminated_length": 77.0,
"epoch": 0.3648,
"grad_norm": 6.931374549865723,
"kl": 7.671875,
"learning_rate": 1e-06,
"loss": 0.6147,
"num_tokens": 5906094.0,
"reward": -11.6011962890625,
"reward_std": 11.895530700683594,
"rewards/rm_reward_func/mean": -11.6011962890625,
"rewards/rm_reward_func/std": 14.706385612487793,
"step": 456
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.53125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 473.0,
"completions/mean_length": 411.875,
"completions/mean_terminated_length": 298.4000244140625,
"completions/min_length": 79.0,
"completions/min_terminated_length": 79.0,
"epoch": 0.3656,
"grad_norm": 7.6649980545043945,
"kl": 6.799072265625,
"learning_rate": 1e-06,
"loss": 0.4754,
"num_tokens": 5921498.0,
"reward": -13.50347900390625,
"reward_std": 9.386650085449219,
"rewards/rm_reward_func/mean": -13.50347900390625,
"rewards/rm_reward_func/std": 14.427614212036133,
"step": 457
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.4375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 475.0,
"completions/mean_length": 412.59375,
"completions/mean_terminated_length": 335.27777099609375,
"completions/min_length": 184.0,
"completions/min_terminated_length": 184.0,
"epoch": 0.3664,
"grad_norm": 14.270245552062988,
"kl": 3.159912109375,
"learning_rate": 1e-06,
"loss": 0.2522,
"num_tokens": 5939749.0,
"reward": 0.822509765625,
"reward_std": 10.554486274719238,
"rewards/rm_reward_func/mean": 0.822509765625,
"rewards/rm_reward_func/std": 19.273405075073242,
"step": 458
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 478.0,
"completions/mean_length": 436.84375,
"completions/mean_terminated_length": 391.75,
"completions/min_length": 145.0,
"completions/min_terminated_length": 145.0,
"epoch": 0.3672,
"grad_norm": 7.888670921325684,
"kl": 3.0789794921875,
"learning_rate": 1e-06,
"loss": 0.1706,
"num_tokens": 5957832.0,
"reward": -1.209869384765625,
"reward_std": 4.854766368865967,
"rewards/rm_reward_func/mean": -1.209869384765625,
"rewards/rm_reward_func/std": 25.843124389648438,
"step": 459
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 477.0,
"completions/mean_length": 439.1875,
"completions/mean_terminated_length": 317.8333435058594,
"completions/min_length": 90.0,
"completions/min_terminated_length": 90.0,
"epoch": 0.368,
"grad_norm": 17.467594146728516,
"kl": 6.8984375,
"learning_rate": 1e-06,
"loss": 0.4475,
"num_tokens": 5975382.0,
"reward": -8.3388671875,
"reward_std": 10.500482559204102,
"rewards/rm_reward_func/mean": -8.3388671875,
"rewards/rm_reward_func/std": 15.440128326416016,
"step": 460
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 373.3125,
"completions/mean_terminated_length": 334.47998046875,
"completions/min_length": 73.0,
"completions/min_terminated_length": 73.0,
"epoch": 0.3688,
"grad_norm": 19.844507217407227,
"kl": 2.7052001953125,
"learning_rate": 1e-06,
"loss": 0.379,
"num_tokens": 5990128.0,
"reward": 8.0970458984375,
"reward_std": 15.225787162780762,
"rewards/rm_reward_func/mean": 8.0970458984375,
"rewards/rm_reward_func/std": 18.16506004333496,
"step": 461
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 507.0,
"completions/mean_length": 446.375,
"completions/mean_terminated_length": 412.0,
"completions/min_length": 232.0,
"completions/min_terminated_length": 232.0,
"epoch": 0.3696,
"grad_norm": 7.17181921005249,
"kl": 0.4322509765625,
"learning_rate": 1e-06,
"loss": 0.0593,
"num_tokens": 6007132.0,
"reward": 8.548095703125,
"reward_std": 7.690432548522949,
"rewards/rm_reward_func/mean": 8.548095703125,
"rewards/rm_reward_func/std": 8.413957595825195,
"step": 462
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.53125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 439.125,
"completions/mean_terminated_length": 356.5333557128906,
"completions/min_length": 42.0,
"completions/min_terminated_length": 42.0,
"epoch": 0.3704,
"grad_norm": 5.347095012664795,
"kl": 6.09375,
"learning_rate": 1e-06,
"loss": 0.3736,
"num_tokens": 6025032.0,
"reward": 1.619140625,
"reward_std": 15.041004180908203,
"rewards/rm_reward_func/mean": 1.619140625,
"rewards/rm_reward_func/std": 28.54711151123047,
"step": 463
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 498.0,
"completions/mean_length": 247.9375,
"completions/mean_terminated_length": 239.41934204101562,
"completions/min_length": 48.0,
"completions/min_terminated_length": 48.0,
"epoch": 0.3712,
"grad_norm": 10.885205268859863,
"kl": 1.0289306640625,
"learning_rate": 1e-06,
"loss": 0.0564,
"num_tokens": 6035230.0,
"reward": 15.275390625,
"reward_std": 8.497432708740234,
"rewards/rm_reward_func/mean": 15.275390625,
"rewards/rm_reward_func/std": 14.570096015930176,
"step": 464
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 474.0,
"completions/mean_length": 345.34375,
"completions/mean_terminated_length": 314.4814758300781,
"completions/min_length": 53.0,
"completions/min_terminated_length": 53.0,
"epoch": 0.372,
"grad_norm": 16.305341720581055,
"kl": 2.215576171875,
"learning_rate": 1e-06,
"loss": 0.2026,
"num_tokens": 6048537.0,
"reward": 1.36273193359375,
"reward_std": 9.692459106445312,
"rewards/rm_reward_func/mean": 1.36273193359375,
"rewards/rm_reward_func/std": 17.84818458557129,
"step": 465
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.5,
"completions/max_length": 512.0,
"completions/max_terminated_length": 499.0,
"completions/mean_length": 407.4375,
"completions/mean_terminated_length": 302.875,
"completions/min_length": 62.0,
"completions/min_terminated_length": 62.0,
"epoch": 0.3728,
"grad_norm": 7.994433879852295,
"kl": 3.80224609375,
"learning_rate": 1e-06,
"loss": 0.2663,
"num_tokens": 6064695.0,
"reward": 3.369873046875,
"reward_std": 8.856344223022461,
"rewards/rm_reward_func/mean": 3.369873046875,
"rewards/rm_reward_func/std": 22.26304817199707,
"step": 466
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 458.0,
"completions/mean_length": 290.5,
"completions/mean_terminated_length": 283.3548278808594,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.3736,
"grad_norm": 6.062528133392334,
"kl": 0.313720703125,
"learning_rate": 1e-06,
"loss": 0.0317,
"num_tokens": 6077287.0,
"reward": 5.29248046875,
"reward_std": 8.455047607421875,
"rewards/rm_reward_func/mean": 5.29248046875,
"rewards/rm_reward_func/std": 13.72189998626709,
"step": 467
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 497.0,
"completions/mean_length": 396.28125,
"completions/mean_terminated_length": 326.8500061035156,
"completions/min_length": 79.0,
"completions/min_terminated_length": 79.0,
"epoch": 0.3744,
"grad_norm": 13.018776893615723,
"kl": 3.998291015625,
"learning_rate": 1e-06,
"loss": 0.3904,
"num_tokens": 6092112.0,
"reward": 0.782562255859375,
"reward_std": 9.454818725585938,
"rewards/rm_reward_func/mean": 0.782562255859375,
"rewards/rm_reward_func/std": 14.342537879943848,
"step": 468
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 504.0,
"completions/mean_length": 457.0,
"completions/mean_terminated_length": 428.19049072265625,
"completions/min_length": 363.0,
"completions/min_terminated_length": 363.0,
"epoch": 0.3752,
"grad_norm": 7.1996026039123535,
"kl": 3.93896484375,
"learning_rate": 1e-06,
"loss": 0.1778,
"num_tokens": 6109216.0,
"reward": 5.27410888671875,
"reward_std": 14.364347457885742,
"rewards/rm_reward_func/mean": 5.27410888671875,
"rewards/rm_reward_func/std": 22.558135986328125,
"step": 469
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.4375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 507.0,
"completions/mean_length": 405.4375,
"completions/mean_terminated_length": 322.5555725097656,
"completions/min_length": 76.0,
"completions/min_terminated_length": 76.0,
"epoch": 0.376,
"grad_norm": 19.028276443481445,
"kl": 8.43359375,
"learning_rate": 1e-06,
"loss": 0.5417,
"num_tokens": 6124742.0,
"reward": -0.470703125,
"reward_std": 12.874818801879883,
"rewards/rm_reward_func/mean": -0.470703125,
"rewards/rm_reward_func/std": 20.95646095275879,
"step": 470
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.40625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 398.0,
"completions/mean_length": 392.8125,
"completions/mean_terminated_length": 311.2631530761719,
"completions/min_length": 217.0,
"completions/min_terminated_length": 217.0,
"epoch": 0.3768,
"grad_norm": 28.145797729492188,
"kl": 6.830322265625,
"learning_rate": 1e-06,
"loss": 0.4005,
"num_tokens": 6141808.0,
"reward": -3.615478515625,
"reward_std": 9.015436172485352,
"rewards/rm_reward_func/mean": -3.615478515625,
"rewards/rm_reward_func/std": 18.243576049804688,
"step": 471
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 475.0,
"completions/mean_length": 417.125,
"completions/mean_terminated_length": 374.0,
"completions/min_length": 62.0,
"completions/min_terminated_length": 62.0,
"epoch": 0.3776,
"grad_norm": 41.16624450683594,
"kl": 7.7880859375,
"learning_rate": 1e-06,
"loss": 0.3244,
"num_tokens": 6157436.0,
"reward": 10.93359375,
"reward_std": 15.283958435058594,
"rewards/rm_reward_func/mean": 10.93359375,
"rewards/rm_reward_func/std": 27.59185028076172,
"step": 472
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 492.0,
"completions/mean_length": 351.3125,
"completions/mean_terminated_length": 278.2727355957031,
"completions/min_length": 30.0,
"completions/min_terminated_length": 30.0,
"epoch": 0.3784,
"grad_norm": 19.67943572998047,
"kl": 7.505615234375,
"learning_rate": 1e-06,
"loss": 0.536,
"num_tokens": 6173246.0,
"reward": 1.380859375,
"reward_std": 14.768939971923828,
"rewards/rm_reward_func/mean": 1.380859375,
"rewards/rm_reward_func/std": 18.74730682373047,
"step": 473
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.40625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 482.0,
"completions/mean_length": 406.28125,
"completions/mean_terminated_length": 333.9473571777344,
"completions/min_length": 67.0,
"completions/min_terminated_length": 67.0,
"epoch": 0.3792,
"grad_norm": 14.080571174621582,
"kl": 6.677734375,
"learning_rate": 1e-06,
"loss": 0.3526,
"num_tokens": 6189295.0,
"reward": -8.592529296875,
"reward_std": 8.344656944274902,
"rewards/rm_reward_func/mean": -8.592529296875,
"rewards/rm_reward_func/std": 14.625808715820312,
"step": 474
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 372.5625,
"completions/mean_terminated_length": 326.0833435058594,
"completions/min_length": 74.0,
"completions/min_terminated_length": 74.0,
"epoch": 0.38,
"grad_norm": 55.990806579589844,
"kl": 3.190673828125,
"learning_rate": 1e-06,
"loss": 0.3699,
"num_tokens": 6204681.0,
"reward": 10.92529296875,
"reward_std": 17.398414611816406,
"rewards/rm_reward_func/mean": 10.92529296875,
"rewards/rm_reward_func/std": 21.6011905670166,
"step": 475
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 473.0,
"completions/mean_length": 336.34375,
"completions/mean_terminated_length": 303.8148193359375,
"completions/min_length": 71.0,
"completions/min_terminated_length": 71.0,
"epoch": 0.3808,
"grad_norm": 26.353031158447266,
"kl": 3.114013671875,
"learning_rate": 1e-06,
"loss": 0.2881,
"num_tokens": 6217324.0,
"reward": -5.37896728515625,
"reward_std": 7.184835433959961,
"rewards/rm_reward_func/mean": -5.37896728515625,
"rewards/rm_reward_func/std": 18.45599365234375,
"step": 476
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 505.0,
"completions/mean_length": 417.75,
"completions/mean_terminated_length": 396.0,
"completions/min_length": 193.0,
"completions/min_terminated_length": 193.0,
"epoch": 0.3816,
"grad_norm": 8.078795433044434,
"kl": 3.979248046875,
"learning_rate": 1e-06,
"loss": 0.2362,
"num_tokens": 6233940.0,
"reward": 8.0015869140625,
"reward_std": 12.765542984008789,
"rewards/rm_reward_func/mean": 8.0015869140625,
"rewards/rm_reward_func/std": 16.40848159790039,
"step": 477
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 393.375,
"completions/mean_terminated_length": 322.20001220703125,
"completions/min_length": 188.0,
"completions/min_terminated_length": 188.0,
"epoch": 0.3824,
"grad_norm": 16.03205108642578,
"kl": 7.90283203125,
"learning_rate": 1e-06,
"loss": 0.3989,
"num_tokens": 6249568.0,
"reward": -4.557373046875,
"reward_std": 9.42959213256836,
"rewards/rm_reward_func/mean": -4.557373046875,
"rewards/rm_reward_func/std": 19.618621826171875,
"step": 478
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 414.0,
"completions/mean_length": 391.4375,
"completions/mean_terminated_length": 328.28570556640625,
"completions/min_length": 165.0,
"completions/min_terminated_length": 165.0,
"epoch": 0.3832,
"grad_norm": 21.212261199951172,
"kl": 7.78076171875,
"learning_rate": 1e-06,
"loss": 0.364,
"num_tokens": 6264534.0,
"reward": 0.4144287109375,
"reward_std": 8.546712875366211,
"rewards/rm_reward_func/mean": 0.4144287109375,
"rewards/rm_reward_func/std": 16.865026473999023,
"step": 479
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 501.0,
"completions/mean_length": 370.65625,
"completions/mean_terminated_length": 323.54168701171875,
"completions/min_length": 219.0,
"completions/min_terminated_length": 219.0,
"epoch": 0.384,
"grad_norm": 25.12664031982422,
"kl": 6.88818359375,
"learning_rate": 1e-06,
"loss": 0.3655,
"num_tokens": 6278235.0,
"reward": 4.96453857421875,
"reward_std": 17.571796417236328,
"rewards/rm_reward_func/mean": 4.96453857421875,
"rewards/rm_reward_func/std": 19.67218589782715,
"step": 480
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.6875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 209.0,
"completions/mean_length": 390.875,
"completions/mean_terminated_length": 124.4000015258789,
"completions/min_length": 70.0,
"completions/min_terminated_length": 70.0,
"epoch": 0.3848,
"grad_norm": 79.35224914550781,
"kl": 21.0,
"learning_rate": 1e-06,
"loss": 1.0686,
"num_tokens": 6294447.0,
"reward": -10.9208984375,
"reward_std": 13.471847534179688,
"rewards/rm_reward_func/mean": -10.9208984375,
"rewards/rm_reward_func/std": 27.3441219329834,
"step": 481
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.6875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 229.0,
"completions/mean_length": 404.0,
"completions/mean_terminated_length": 166.40000915527344,
"completions/min_length": 38.0,
"completions/min_terminated_length": 38.0,
"epoch": 0.3856,
"grad_norm": 132.53387451171875,
"kl": 13.1297607421875,
"learning_rate": 1e-06,
"loss": 0.7436,
"num_tokens": 6311167.0,
"reward": -12.4658203125,
"reward_std": 8.332673072814941,
"rewards/rm_reward_func/mean": -12.4658203125,
"rewards/rm_reward_func/std": 14.378026962280273,
"step": 482
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.46875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 486.0,
"completions/mean_length": 379.8125,
"completions/mean_terminated_length": 263.1764831542969,
"completions/min_length": 59.0,
"completions/min_terminated_length": 59.0,
"epoch": 0.3864,
"grad_norm": 17.89412498474121,
"kl": 7.271484375,
"learning_rate": 1e-06,
"loss": 0.4685,
"num_tokens": 6325729.0,
"reward": -2.525604248046875,
"reward_std": 10.174657821655273,
"rewards/rm_reward_func/mean": -2.525604248046875,
"rewards/rm_reward_func/std": 19.22846221923828,
"step": 483
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 355.34375,
"completions/mean_terminated_length": 326.3333435058594,
"completions/min_length": 93.0,
"completions/min_terminated_length": 93.0,
"epoch": 0.3872,
"grad_norm": 10.339418411254883,
"kl": 5.09521484375,
"learning_rate": 1e-06,
"loss": 0.349,
"num_tokens": 6339132.0,
"reward": 0.1297607421875,
"reward_std": 12.494827270507812,
"rewards/rm_reward_func/mean": 0.1297607421875,
"rewards/rm_reward_func/std": 17.290435791015625,
"step": 484
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.59375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 388.0,
"completions/mean_length": 372.75,
"completions/mean_terminated_length": 169.23077392578125,
"completions/min_length": 72.0,
"completions/min_terminated_length": 72.0,
"epoch": 0.388,
"grad_norm": 12.62641429901123,
"kl": 8.357177734375,
"learning_rate": 1e-06,
"loss": 0.6359,
"num_tokens": 6357132.0,
"reward": -7.014617919921875,
"reward_std": 11.388704299926758,
"rewards/rm_reward_func/mean": -7.014617919921875,
"rewards/rm_reward_func/std": 11.901640892028809,
"step": 485
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 450.0,
"completions/mean_length": 307.125,
"completions/mean_terminated_length": 214.0,
"completions/min_length": 57.0,
"completions/min_terminated_length": 57.0,
"epoch": 0.3888,
"grad_norm": 13.34914779663086,
"kl": 6.413330078125,
"learning_rate": 1e-06,
"loss": 0.5975,
"num_tokens": 6369224.0,
"reward": -6.581207275390625,
"reward_std": 14.152790069580078,
"rewards/rm_reward_func/mean": -6.581207275390625,
"rewards/rm_reward_func/std": 17.590269088745117,
"step": 486
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 476.0,
"completions/mean_length": 422.0625,
"completions/mean_terminated_length": 368.1000061035156,
"completions/min_length": 241.0,
"completions/min_terminated_length": 241.0,
"epoch": 0.3896,
"grad_norm": 15.647866249084473,
"kl": 5.72802734375,
"learning_rate": 1e-06,
"loss": 0.3176,
"num_tokens": 6385186.0,
"reward": 2.65093994140625,
"reward_std": 8.202743530273438,
"rewards/rm_reward_func/mean": 2.65093994140625,
"rewards/rm_reward_func/std": 15.850854873657227,
"step": 487
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 359.40625,
"completions/mean_terminated_length": 279.4761962890625,
"completions/min_length": 72.0,
"completions/min_terminated_length": 72.0,
"epoch": 0.3904,
"grad_norm": 43.07307434082031,
"kl": 6.449462890625,
"learning_rate": 1e-06,
"loss": 0.5931,
"num_tokens": 6400463.0,
"reward": -1.673583984375,
"reward_std": 10.083917617797852,
"rewards/rm_reward_func/mean": -1.673583984375,
"rewards/rm_reward_func/std": 13.745939254760742,
"step": 488
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.5,
"completions/max_length": 512.0,
"completions/max_terminated_length": 394.0,
"completions/mean_length": 399.375,
"completions/mean_terminated_length": 286.75,
"completions/min_length": 110.0,
"completions/min_terminated_length": 110.0,
"epoch": 0.3912,
"grad_norm": 91.3353042602539,
"kl": 7.009765625,
"learning_rate": 1e-06,
"loss": 0.4609,
"num_tokens": 6415571.0,
"reward": -7.83154296875,
"reward_std": 9.295122146606445,
"rewards/rm_reward_func/mean": -7.83154296875,
"rewards/rm_reward_func/std": 16.444271087646484,
"step": 489
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 484.0,
"completions/mean_length": 362.71875,
"completions/mean_terminated_length": 328.2692565917969,
"completions/min_length": 133.0,
"completions/min_terminated_length": 133.0,
"epoch": 0.392,
"grad_norm": 24.025558471679688,
"kl": 3.818359375,
"learning_rate": 1e-06,
"loss": 0.3076,
"num_tokens": 6429994.0,
"reward": 12.959228515625,
"reward_std": 14.361933708190918,
"rewards/rm_reward_func/mean": 12.959228515625,
"rewards/rm_reward_func/std": 26.099708557128906,
"step": 490
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.5,
"completions/max_length": 512.0,
"completions/max_terminated_length": 419.0,
"completions/mean_length": 348.15625,
"completions/mean_terminated_length": 184.3125,
"completions/min_length": 81.0,
"completions/min_terminated_length": 81.0,
"epoch": 0.3928,
"grad_norm": 49.87342071533203,
"kl": 9.898193359375,
"learning_rate": 1e-06,
"loss": 0.8155,
"num_tokens": 6449151.0,
"reward": -8.174560546875,
"reward_std": 11.404885292053223,
"rewards/rm_reward_func/mean": -8.174560546875,
"rewards/rm_reward_func/std": 15.167132377624512,
"step": 491
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 502.0,
"completions/mean_length": 308.5,
"completions/mean_terminated_length": 270.8148193359375,
"completions/min_length": 71.0,
"completions/min_terminated_length": 71.0,
"epoch": 0.3936,
"grad_norm": 42.138450622558594,
"kl": 4.73828125,
"learning_rate": 1e-06,
"loss": 0.5528,
"num_tokens": 6463335.0,
"reward": 11.81298828125,
"reward_std": 15.574457168579102,
"rewards/rm_reward_func/mean": 11.81298828125,
"rewards/rm_reward_func/std": 20.5043888092041,
"step": 492
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 486.0,
"completions/mean_length": 285.34375,
"completions/mean_terminated_length": 166.61904907226562,
"completions/min_length": 79.0,
"completions/min_terminated_length": 79.0,
"epoch": 0.3944,
"grad_norm": 122.83219146728516,
"kl": 9.7734375,
"learning_rate": 1e-06,
"loss": 0.9872,
"num_tokens": 6475618.0,
"reward": -2.90985107421875,
"reward_std": 13.12482738494873,
"rewards/rm_reward_func/mean": -2.90985107421875,
"rewards/rm_reward_func/std": 14.76265811920166,
"step": 493
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 497.0,
"completions/mean_length": 352.0625,
"completions/mean_terminated_length": 256.1000061035156,
"completions/min_length": 82.0,
"completions/min_terminated_length": 82.0,
"epoch": 0.3952,
"grad_norm": 18.307031631469727,
"kl": 11.32470703125,
"learning_rate": 1e-06,
"loss": 0.8083,
"num_tokens": 6489324.0,
"reward": -12.8271484375,
"reward_std": 8.004693984985352,
"rewards/rm_reward_func/mean": -12.8271484375,
"rewards/rm_reward_func/std": 11.452200889587402,
"step": 494
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 499.0,
"completions/mean_length": 361.25,
"completions/mean_terminated_length": 302.2608642578125,
"completions/min_length": 64.0,
"completions/min_terminated_length": 64.0,
"epoch": 0.396,
"grad_norm": 24.6053409576416,
"kl": 6.4337158203125,
"learning_rate": 1e-06,
"loss": 0.2939,
"num_tokens": 6507252.0,
"reward": -5.3721923828125,
"reward_std": 7.6330060958862305,
"rewards/rm_reward_func/mean": -5.3721923828125,
"rewards/rm_reward_func/std": 12.24431324005127,
"step": 495
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 278.3125,
"completions/mean_terminated_length": 200.4166717529297,
"completions/min_length": 66.0,
"completions/min_terminated_length": 66.0,
"epoch": 0.3968,
"grad_norm": 15.430408477783203,
"kl": 4.6640625,
"learning_rate": 1e-06,
"loss": 0.4849,
"num_tokens": 6521006.0,
"reward": 11.46609115600586,
"reward_std": 17.760250091552734,
"rewards/rm_reward_func/mean": 11.46609115600586,
"rewards/rm_reward_func/std": 25.725770950317383,
"step": 496
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.6875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 490.0,
"completions/mean_length": 470.03125,
"completions/mean_terminated_length": 377.70001220703125,
"completions/min_length": 75.0,
"completions/min_terminated_length": 75.0,
"epoch": 0.3976,
"grad_norm": 171.98805236816406,
"kl": 15.0703125,
"learning_rate": 1e-06,
"loss": 0.7248,
"num_tokens": 6539287.0,
"reward": -14.515625,
"reward_std": 12.340795516967773,
"rewards/rm_reward_func/mean": -14.515625,
"rewards/rm_reward_func/std": 17.410518646240234,
"step": 497
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.6875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 479.625,
"completions/mean_terminated_length": 408.3999938964844,
"completions/min_length": 275.0,
"completions/min_terminated_length": 275.0,
"epoch": 0.3984,
"grad_norm": 38.708248138427734,
"kl": 13.7109375,
"learning_rate": 1e-06,
"loss": 0.6117,
"num_tokens": 6556987.0,
"reward": -11.59375,
"reward_std": 13.794965744018555,
"rewards/rm_reward_func/mean": -11.59375,
"rewards/rm_reward_func/std": 23.78890609741211,
"step": 498
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 421.0,
"completions/mean_length": 409.3125,
"completions/mean_terminated_length": 238.1666717529297,
"completions/min_length": 77.0,
"completions/min_terminated_length": 77.0,
"epoch": 0.3992,
"grad_norm": 25.57811164855957,
"kl": 14.84375,
"learning_rate": 1e-06,
"loss": 0.8756,
"num_tokens": 6572293.0,
"reward": -6.9130859375,
"reward_std": 14.296964645385742,
"rewards/rm_reward_func/mean": -6.9130859375,
"rewards/rm_reward_func/std": 21.80596351623535,
"step": 499
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.4375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 485.0,
"completions/mean_length": 378.0,
"completions/mean_terminated_length": 273.77777099609375,
"completions/min_length": 70.0,
"completions/min_terminated_length": 70.0,
"epoch": 0.4,
"grad_norm": 70.98407745361328,
"kl": 7.545654296875,
"learning_rate": 1e-06,
"loss": 0.5359,
"num_tokens": 6586981.0,
"reward": -2.8328094482421875,
"reward_std": 7.248239040374756,
"rewards/rm_reward_func/mean": -2.8328094482421875,
"rewards/rm_reward_func/std": 12.194914817810059,
"step": 500
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.8125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 496.78125,
"completions/mean_terminated_length": 430.8333435058594,
"completions/min_length": 373.0,
"completions/min_terminated_length": 373.0,
"epoch": 0.4008,
"grad_norm": 9.479290008544922,
"kl": 6.978515625,
"learning_rate": 1e-06,
"loss": 0.3081,
"num_tokens": 6606366.0,
"reward": -7.8212890625,
"reward_std": 11.259090423583984,
"rewards/rm_reward_func/mean": -7.8212890625,
"rewards/rm_reward_func/std": 18.525226593017578,
"step": 501
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.46875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 385.9375,
"completions/mean_terminated_length": 274.70587158203125,
"completions/min_length": 67.0,
"completions/min_terminated_length": 67.0,
"epoch": 0.4016,
"grad_norm": 42.01311492919922,
"kl": 5.5234375,
"learning_rate": 1e-06,
"loss": 0.5541,
"num_tokens": 6621700.0,
"reward": -0.1533203125,
"reward_std": 14.605756759643555,
"rewards/rm_reward_func/mean": -0.1533203125,
"rewards/rm_reward_func/std": 23.270185470581055,
"step": 502
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.71875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 454.0,
"completions/mean_length": 436.625,
"completions/mean_terminated_length": 244.0,
"completions/min_length": 61.0,
"completions/min_terminated_length": 61.0,
"epoch": 0.4024,
"grad_norm": 20.074079513549805,
"kl": 7.421875,
"learning_rate": 1e-06,
"loss": 0.5378,
"num_tokens": 6637896.0,
"reward": -12.885040283203125,
"reward_std": 12.589856147766113,
"rewards/rm_reward_func/mean": -12.885040283203125,
"rewards/rm_reward_func/std": 16.110185623168945,
"step": 503
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 231.0,
"completions/mean_length": 243.875,
"completions/mean_terminated_length": 103.42857360839844,
"completions/min_length": 58.0,
"completions/min_terminated_length": 58.0,
"epoch": 0.4032,
"grad_norm": 24.548263549804688,
"kl": 8.2744140625,
"learning_rate": 1e-06,
"loss": 0.9126,
"num_tokens": 6648780.0,
"reward": -6.2499847412109375,
"reward_std": 12.00403118133545,
"rewards/rm_reward_func/mean": -6.2499847412109375,
"rewards/rm_reward_func/std": 13.340556144714355,
"step": 504
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.59375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 490.0,
"completions/mean_length": 460.0625,
"completions/mean_terminated_length": 384.15386962890625,
"completions/min_length": 214.0,
"completions/min_terminated_length": 214.0,
"epoch": 0.404,
"grad_norm": 17.74894905090332,
"kl": 4.763671875,
"learning_rate": 1e-06,
"loss": 0.2654,
"num_tokens": 6666302.0,
"reward": -6.497314453125,
"reward_std": 12.197681427001953,
"rewards/rm_reward_func/mean": -6.497314453125,
"rewards/rm_reward_func/std": 15.69522476196289,
"step": 505
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.5625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 426.0,
"completions/mean_length": 376.0,
"completions/mean_terminated_length": 201.1428680419922,
"completions/min_length": 75.0,
"completions/min_terminated_length": 75.0,
"epoch": 0.4048,
"grad_norm": 13.425418853759766,
"kl": 10.6875,
"learning_rate": 1e-06,
"loss": 0.8323,
"num_tokens": 6683206.0,
"reward": -5.591552734375,
"reward_std": 16.38616943359375,
"rewards/rm_reward_func/mean": -5.591552734375,
"rewards/rm_reward_func/std": 19.70439910888672,
"step": 506
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.6875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 501.0,
"completions/mean_length": 437.09375,
"completions/mean_terminated_length": 272.3000183105469,
"completions/min_length": 64.0,
"completions/min_terminated_length": 64.0,
"epoch": 0.4056,
"grad_norm": 23.703475952148438,
"kl": 9.7734375,
"learning_rate": 1e-06,
"loss": 0.5968,
"num_tokens": 6700297.0,
"reward": -4.63623046875,
"reward_std": 19.530458450317383,
"rewards/rm_reward_func/mean": -4.63623046875,
"rewards/rm_reward_func/std": 19.052122116088867,
"step": 507
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.46875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 435.0,
"completions/mean_length": 324.25,
"completions/mean_terminated_length": 158.58824157714844,
"completions/min_length": 73.0,
"completions/min_terminated_length": 73.0,
"epoch": 0.4064,
"grad_norm": 28.222517013549805,
"kl": 9.54296875,
"learning_rate": 1e-06,
"loss": 0.5761,
"num_tokens": 6715521.0,
"reward": -10.552978515625,
"reward_std": 8.230770111083984,
"rewards/rm_reward_func/mean": -10.552978515625,
"rewards/rm_reward_func/std": 15.682586669921875,
"step": 508
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 1.0,
"completions/max_length": 512.0,
"completions/max_terminated_length": 0.0,
"completions/mean_length": 512.0,
"completions/mean_terminated_length": 0.0,
"completions/min_length": 512.0,
"completions/min_terminated_length": 0.0,
"epoch": 0.4072,
"grad_norm": 38.62055969238281,
"kl": 15.5,
"learning_rate": 1e-06,
"loss": 0.6205,
"num_tokens": 6733761.0,
"reward": -21.97564697265625,
"reward_std": 5.978219985961914,
"rewards/rm_reward_func/mean": -21.97564697265625,
"rewards/rm_reward_func/std": 7.307789325714111,
"step": 509
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.4375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 366.53125,
"completions/mean_terminated_length": 253.38888549804688,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.408,
"grad_norm": 25.734407424926758,
"kl": 8.80078125,
"learning_rate": 1e-06,
"loss": 0.6397,
"num_tokens": 6749586.0,
"reward": -2.8720703125,
"reward_std": 11.287141799926758,
"rewards/rm_reward_func/mean": -2.8720703125,
"rewards/rm_reward_func/std": 18.167095184326172,
"step": 510
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.59375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 449.0,
"completions/mean_length": 436.15625,
"completions/mean_terminated_length": 325.3077087402344,
"completions/min_length": 231.0,
"completions/min_terminated_length": 231.0,
"epoch": 0.4088,
"grad_norm": 21.79092025756836,
"kl": 8.068603515625,
"learning_rate": 1e-06,
"loss": 0.3422,
"num_tokens": 6765663.0,
"reward": -13.787574768066406,
"reward_std": 6.30387020111084,
"rewards/rm_reward_func/mean": -13.787574768066406,
"rewards/rm_reward_func/std": 17.804723739624023,
"step": 511
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.71875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 487.0,
"completions/mean_length": 420.71875,
"completions/mean_terminated_length": 187.44444274902344,
"completions/min_length": 79.0,
"completions/min_terminated_length": 79.0,
"epoch": 0.4096,
"grad_norm": 161.0446319580078,
"kl": 9.1796875,
"learning_rate": 1e-06,
"loss": 0.6176,
"num_tokens": 6782334.0,
"reward": -11.8310546875,
"reward_std": 9.512861251831055,
"rewards/rm_reward_func/mean": -11.8310546875,
"rewards/rm_reward_func/std": 16.280780792236328,
"step": 512
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.59375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 397.84375,
"completions/mean_terminated_length": 231.00001525878906,
"completions/min_length": 69.0,
"completions/min_terminated_length": 69.0,
"epoch": 0.4104,
"grad_norm": 12.210495948791504,
"kl": 4.85791015625,
"learning_rate": 1e-06,
"loss": 0.4964,
"num_tokens": 6798361.0,
"reward": 0.423828125,
"reward_std": 10.785526275634766,
"rewards/rm_reward_func/mean": 0.423828125,
"rewards/rm_reward_func/std": 16.22878074645996,
"step": 513
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.40625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 511.0,
"completions/mean_length": 384.9375,
"completions/mean_terminated_length": 298.0,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.4112,
"grad_norm": 27.765371322631836,
"kl": 5.994140625,
"learning_rate": 1e-06,
"loss": 0.5379,
"num_tokens": 6813559.0,
"reward": -0.2978515625,
"reward_std": 15.280010223388672,
"rewards/rm_reward_func/mean": -0.2978515625,
"rewards/rm_reward_func/std": 18.968994140625,
"step": 514
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.75,
"completions/max_length": 512.0,
"completions/max_terminated_length": 373.0,
"completions/mean_length": 456.125,
"completions/mean_terminated_length": 288.5,
"completions/min_length": 67.0,
"completions/min_terminated_length": 67.0,
"epoch": 0.412,
"grad_norm": 17.61673927307129,
"kl": 7.875,
"learning_rate": 1e-06,
"loss": 0.4006,
"num_tokens": 6832075.0,
"reward": -14.86328125,
"reward_std": 6.095500946044922,
"rewards/rm_reward_func/mean": -14.86328125,
"rewards/rm_reward_func/std": 16.34231948852539,
"step": 515
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.8125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 488.0,
"completions/mean_length": 480.84375,
"completions/mean_terminated_length": 345.8333435058594,
"completions/min_length": 73.0,
"completions/min_terminated_length": 73.0,
"epoch": 0.4128,
"grad_norm": 17.899005889892578,
"kl": 6.27734375,
"learning_rate": 1e-06,
"loss": 0.2511,
"num_tokens": 6849974.0,
"reward": -14.375091552734375,
"reward_std": 6.907338619232178,
"rewards/rm_reward_func/mean": -14.375091552734375,
"rewards/rm_reward_func/std": 9.020212173461914,
"step": 516
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.5625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 268.0,
"completions/mean_length": 378.34375,
"completions/mean_terminated_length": 206.50001525878906,
"completions/min_length": 158.0,
"completions/min_terminated_length": 158.0,
"epoch": 0.4136,
"grad_norm": 25.587743759155273,
"kl": 5.07421875,
"learning_rate": 1e-06,
"loss": 0.3087,
"num_tokens": 6866065.0,
"reward": -9.5302734375,
"reward_std": 5.052003860473633,
"rewards/rm_reward_func/mean": -9.5302734375,
"rewards/rm_reward_func/std": 10.77525520324707,
"step": 517
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.59375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 491.0,
"completions/mean_length": 383.625,
"completions/mean_terminated_length": 196.0,
"completions/min_length": 79.0,
"completions/min_terminated_length": 79.0,
"epoch": 0.4144,
"grad_norm": 28.90671730041504,
"kl": 4.85546875,
"learning_rate": 1e-06,
"loss": 0.5353,
"num_tokens": 6883749.0,
"reward": -7.447265625,
"reward_std": 11.519471168518066,
"rewards/rm_reward_func/mean": -7.447265625,
"rewards/rm_reward_func/std": 13.736445426940918,
"step": 518
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.71875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 504.0,
"completions/mean_length": 460.6875,
"completions/mean_terminated_length": 329.5555725097656,
"completions/min_length": 205.0,
"completions/min_terminated_length": 205.0,
"epoch": 0.4152,
"grad_norm": 6.363600254058838,
"kl": 5.875,
"learning_rate": 1e-06,
"loss": 0.3459,
"num_tokens": 6901867.0,
"reward": -8.869140625,
"reward_std": 12.279427528381348,
"rewards/rm_reward_func/mean": -8.869140625,
"rewards/rm_reward_func/std": 18.361650466918945,
"step": 519
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.5625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 483.0,
"completions/mean_length": 426.90625,
"completions/mean_terminated_length": 317.5,
"completions/min_length": 52.0,
"completions/min_terminated_length": 52.0,
"epoch": 0.416,
"grad_norm": 12.628585815429688,
"kl": 3.195068359375,
"learning_rate": 1e-06,
"loss": 0.2452,
"num_tokens": 6918720.0,
"reward": -3.0980224609375,
"reward_std": 8.710382461547852,
"rewards/rm_reward_func/mean": -3.0980224609375,
"rewards/rm_reward_func/std": 11.913897514343262,
"step": 520
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 423.0,
"completions/mean_length": 296.28125,
"completions/mean_terminated_length": 166.85000610351562,
"completions/min_length": 70.0,
"completions/min_terminated_length": 70.0,
"epoch": 0.4168,
"grad_norm": 17.954702377319336,
"kl": 5.048828125,
"learning_rate": 1e-06,
"loss": 0.4875,
"num_tokens": 6931665.0,
"reward": 0.85546875,
"reward_std": 10.385427474975586,
"rewards/rm_reward_func/mean": 0.85546875,
"rewards/rm_reward_func/std": 15.259444236755371,
"step": 521
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.5625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 447.0,
"completions/mean_length": 409.40625,
"completions/mean_terminated_length": 277.5,
"completions/min_length": 115.0,
"completions/min_terminated_length": 115.0,
"epoch": 0.4176,
"grad_norm": 19.831439971923828,
"kl": 6.078125,
"learning_rate": 1e-06,
"loss": 0.4218,
"num_tokens": 6946854.0,
"reward": -6.93914794921875,
"reward_std": 13.028332710266113,
"rewards/rm_reward_func/mean": -6.93914794921875,
"rewards/rm_reward_func/std": 14.581646919250488,
"step": 522
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.5,
"completions/max_length": 512.0,
"completions/max_terminated_length": 488.0,
"completions/mean_length": 438.09375,
"completions/mean_terminated_length": 364.1875,
"completions/min_length": 181.0,
"completions/min_terminated_length": 181.0,
"epoch": 0.4184,
"grad_norm": 30.67502212524414,
"kl": 4.91064453125,
"learning_rate": 1e-06,
"loss": 0.2753,
"num_tokens": 6963185.0,
"reward": -6.296905517578125,
"reward_std": 5.502427101135254,
"rewards/rm_reward_func/mean": -6.296905517578125,
"rewards/rm_reward_func/std": 12.138696670532227,
"step": 523
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.5625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 322.0,
"completions/mean_length": 346.78125,
"completions/mean_terminated_length": 134.35714721679688,
"completions/min_length": 81.0,
"completions/min_terminated_length": 81.0,
"epoch": 0.4192,
"grad_norm": 33.20919418334961,
"kl": 6.63623046875,
"learning_rate": 1e-06,
"loss": 0.4747,
"num_tokens": 6976906.0,
"reward": -11.229736328125,
"reward_std": 4.540736198425293,
"rewards/rm_reward_func/mean": -11.229736328125,
"rewards/rm_reward_func/std": 11.639115333557129,
"step": 524
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 370.03125,
"completions/mean_terminated_length": 343.7407531738281,
"completions/min_length": 144.0,
"completions/min_terminated_length": 144.0,
"epoch": 0.42,
"grad_norm": 9.042720794677734,
"kl": 0.87451171875,
"learning_rate": 1e-06,
"loss": 0.174,
"num_tokens": 6990763.0,
"reward": 4.61993408203125,
"reward_std": 7.443538665771484,
"rewards/rm_reward_func/mean": 4.61993408203125,
"rewards/rm_reward_func/std": 12.067349433898926,
"step": 525
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.6875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 498.0,
"completions/mean_length": 399.65625,
"completions/mean_terminated_length": 152.5,
"completions/min_length": 52.0,
"completions/min_terminated_length": 52.0,
"epoch": 0.4208,
"grad_norm": 23.3360538482666,
"kl": 12.09375,
"learning_rate": 1e-06,
"loss": 0.8084,
"num_tokens": 7008024.0,
"reward": -11.912353515625,
"reward_std": 9.951939582824707,
"rewards/rm_reward_func/mean": -11.912353515625,
"rewards/rm_reward_func/std": 15.001555442810059,
"step": 526
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 479.0,
"completions/mean_length": 364.9375,
"completions/mean_terminated_length": 287.9047546386719,
"completions/min_length": 81.0,
"completions/min_terminated_length": 81.0,
"epoch": 0.4216,
"grad_norm": 11.807538986206055,
"kl": 3.63671875,
"learning_rate": 1e-06,
"loss": 0.2078,
"num_tokens": 7023238.0,
"reward": 1.132568359375,
"reward_std": 11.639840126037598,
"rewards/rm_reward_func/mean": 1.132568359375,
"rewards/rm_reward_func/std": 15.697065353393555,
"step": 527
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.4375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 325.0,
"completions/mean_length": 308.46875,
"completions/mean_terminated_length": 150.1666717529297,
"completions/min_length": 70.0,
"completions/min_terminated_length": 70.0,
"epoch": 0.4224,
"grad_norm": 10.746874809265137,
"kl": 7.41259765625,
"learning_rate": 1e-06,
"loss": 0.4985,
"num_tokens": 7035693.0,
"reward": -4.9210205078125,
"reward_std": 7.971255302429199,
"rewards/rm_reward_func/mean": -4.9210205078125,
"rewards/rm_reward_func/std": 17.77606773376465,
"step": 528
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.59375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 475.0,
"completions/mean_length": 450.3125,
"completions/mean_terminated_length": 360.15386962890625,
"completions/min_length": 290.0,
"completions/min_terminated_length": 290.0,
"epoch": 0.4232,
"grad_norm": 18.503520965576172,
"kl": 9.25390625,
"learning_rate": 1e-06,
"loss": 0.4482,
"num_tokens": 7054207.0,
"reward": 1.451171875,
"reward_std": 22.573057174682617,
"rewards/rm_reward_func/mean": 1.451171875,
"rewards/rm_reward_func/std": 27.66277313232422,
"step": 529
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.84375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 328.0,
"completions/mean_length": 467.78125,
"completions/mean_terminated_length": 229.0,
"completions/min_length": 70.0,
"completions/min_terminated_length": 70.0,
"epoch": 0.424,
"grad_norm": 47.28842544555664,
"kl": 14.5625,
"learning_rate": 1e-06,
"loss": 0.7504,
"num_tokens": 7072056.0,
"reward": -19.1044921875,
"reward_std": 12.893132209777832,
"rewards/rm_reward_func/mean": -19.1044921875,
"rewards/rm_reward_func/std": 14.44466781616211,
"step": 530
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.8125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 424.0,
"completions/mean_length": 474.28125,
"completions/mean_terminated_length": 310.8333435058594,
"completions/min_length": 229.0,
"completions/min_terminated_length": 229.0,
"epoch": 0.4248,
"grad_norm": 44.54637908935547,
"kl": 8.7578125,
"learning_rate": 1e-06,
"loss": 0.4787,
"num_tokens": 7089625.0,
"reward": -19.9476318359375,
"reward_std": 8.834146499633789,
"rewards/rm_reward_func/mean": -19.9476318359375,
"rewards/rm_reward_func/std": 10.349533081054688,
"step": 531
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.5625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 487.0,
"completions/mean_length": 366.28125,
"completions/mean_terminated_length": 178.92857360839844,
"completions/min_length": 71.0,
"completions/min_terminated_length": 71.0,
"epoch": 0.4256,
"grad_norm": 522.36865234375,
"kl": 6.591796875,
"learning_rate": 1e-06,
"loss": 0.6736,
"num_tokens": 7105522.0,
"reward": -1.458984375,
"reward_std": 9.65514850616455,
"rewards/rm_reward_func/mean": -1.458984375,
"rewards/rm_reward_func/std": 23.94460678100586,
"step": 532
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 390.0,
"completions/mean_length": 395.84375,
"completions/mean_terminated_length": 202.25,
"completions/min_length": 79.0,
"completions/min_terminated_length": 79.0,
"epoch": 0.4264,
"grad_norm": 39.247413635253906,
"kl": 7.3671875,
"learning_rate": 1e-06,
"loss": 0.569,
"num_tokens": 7120965.0,
"reward": -13.78955078125,
"reward_std": 11.25367546081543,
"rewards/rm_reward_func/mean": -13.78955078125,
"rewards/rm_reward_func/std": 15.153644561767578,
"step": 533
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.59375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 474.0,
"completions/mean_length": 418.6875,
"completions/mean_terminated_length": 282.3077087402344,
"completions/min_length": 73.0,
"completions/min_terminated_length": 73.0,
"epoch": 0.4272,
"grad_norm": 21.322107315063477,
"kl": 4.0390625,
"learning_rate": 1e-06,
"loss": 0.2022,
"num_tokens": 7136419.0,
"reward": -12.512939453125,
"reward_std": 14.791849136352539,
"rewards/rm_reward_func/mean": -12.512939453125,
"rewards/rm_reward_func/std": 16.32183074951172,
"step": 534
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.6875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 441.0,
"completions/mean_length": 434.34375,
"completions/mean_terminated_length": 263.5,
"completions/min_length": 75.0,
"completions/min_terminated_length": 75.0,
"epoch": 0.428,
"grad_norm": 18.7730770111084,
"kl": 5.703125,
"learning_rate": 1e-06,
"loss": 0.4185,
"num_tokens": 7152758.0,
"reward": -8.214715957641602,
"reward_std": 9.016678810119629,
"rewards/rm_reward_func/mean": -8.214715957641602,
"rewards/rm_reward_func/std": 10.963976860046387,
"step": 535
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 449.0,
"completions/mean_length": 364.15625,
"completions/mean_terminated_length": 275.45001220703125,
"completions/min_length": 74.0,
"completions/min_terminated_length": 74.0,
"epoch": 0.4288,
"grad_norm": 440.05767822265625,
"kl": 5.39599609375,
"learning_rate": 1e-06,
"loss": 0.5068,
"num_tokens": 7167107.0,
"reward": -0.919921875,
"reward_std": 10.649211883544922,
"rewards/rm_reward_func/mean": -0.919921875,
"rewards/rm_reward_func/std": 23.648954391479492,
"step": 536
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.53125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 404.0,
"completions/mean_length": 369.59375,
"completions/mean_terminated_length": 208.20001220703125,
"completions/min_length": 81.0,
"completions/min_terminated_length": 81.0,
"epoch": 0.4296,
"grad_norm": 12.493600845336914,
"kl": 5.7001953125,
"learning_rate": 1e-06,
"loss": 0.29,
"num_tokens": 7181822.0,
"reward": -3.5146484375,
"reward_std": 7.881195068359375,
"rewards/rm_reward_func/mean": -3.5146484375,
"rewards/rm_reward_func/std": 15.348002433776855,
"step": 537
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.65625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 305.0,
"completions/mean_length": 395.65625,
"completions/mean_terminated_length": 173.5454559326172,
"completions/min_length": 68.0,
"completions/min_terminated_length": 68.0,
"epoch": 0.4304,
"grad_norm": 57.40653991699219,
"kl": 9.09375,
"learning_rate": 1e-06,
"loss": 0.7079,
"num_tokens": 7201579.0,
"reward": -9.9658203125,
"reward_std": 12.702253341674805,
"rewards/rm_reward_func/mean": -9.9658203125,
"rewards/rm_reward_func/std": 15.439082145690918,
"step": 538
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.59375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 296.0,
"completions/mean_length": 341.8125,
"completions/mean_terminated_length": 93.0769271850586,
"completions/min_length": 40.0,
"completions/min_terminated_length": 40.0,
"epoch": 0.4312,
"grad_norm": 15.874788284301758,
"kl": 9.1171875,
"learning_rate": 1e-06,
"loss": 0.6786,
"num_tokens": 7218213.0,
"reward": -14.662109375,
"reward_std": 6.063241004943848,
"rewards/rm_reward_func/mean": -14.662109375,
"rewards/rm_reward_func/std": 12.366585731506348,
"step": 539
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.84375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 499.0,
"completions/mean_length": 466.03125,
"completions/mean_terminated_length": 217.8000030517578,
"completions/min_length": 76.0,
"completions/min_terminated_length": 76.0,
"epoch": 0.432,
"grad_norm": 16.715585708618164,
"kl": 8.3671875,
"learning_rate": 1e-06,
"loss": 0.3327,
"num_tokens": 7235414.0,
"reward": -15.6708984375,
"reward_std": 11.589529991149902,
"rewards/rm_reward_func/mean": -15.6708984375,
"rewards/rm_reward_func/std": 12.419350624084473,
"step": 540
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.53125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 481.0,
"completions/mean_length": 445.1875,
"completions/mean_terminated_length": 369.4666748046875,
"completions/min_length": 258.0,
"completions/min_terminated_length": 258.0,
"epoch": 0.4328,
"grad_norm": 5.2816667556762695,
"kl": 4.4296875,
"learning_rate": 1e-06,
"loss": 0.2283,
"num_tokens": 7252700.0,
"reward": -0.094482421875,
"reward_std": 17.288135528564453,
"rewards/rm_reward_func/mean": -0.094482421875,
"rewards/rm_reward_func/std": 24.340282440185547,
"step": 541
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.40625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 476.0,
"completions/mean_length": 378.9375,
"completions/mean_terminated_length": 287.8947448730469,
"completions/min_length": 100.0,
"completions/min_terminated_length": 100.0,
"epoch": 0.4336,
"grad_norm": 20.250680923461914,
"kl": 4.130859375,
"learning_rate": 1e-06,
"loss": 0.4542,
"num_tokens": 7267650.0,
"reward": -4.664306640625,
"reward_std": 12.11115837097168,
"rewards/rm_reward_func/mean": -4.664306640625,
"rewards/rm_reward_func/std": 14.438556671142578,
"step": 542
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.65625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 300.0,
"completions/mean_length": 387.53125,
"completions/mean_terminated_length": 149.90908813476562,
"completions/min_length": 75.0,
"completions/min_terminated_length": 75.0,
"epoch": 0.4344,
"grad_norm": 24.06378173828125,
"kl": 4.3828125,
"learning_rate": 1e-06,
"loss": 0.4955,
"num_tokens": 7282563.0,
"reward": -3.005126953125,
"reward_std": 11.770925521850586,
"rewards/rm_reward_func/mean": -3.005126953125,
"rewards/rm_reward_func/std": 15.770807266235352,
"step": 543
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 264.90625,
"completions/mean_terminated_length": 168.21739196777344,
"completions/min_length": 26.0,
"completions/min_terminated_length": 26.0,
"epoch": 0.4352,
"grad_norm": 27.67669677734375,
"kl": 3.830078125,
"learning_rate": 1e-06,
"loss": 0.5483,
"num_tokens": 7293816.0,
"reward": -2.7064208984375,
"reward_std": 13.272294998168945,
"rewards/rm_reward_func/mean": -2.7064208984375,
"rewards/rm_reward_func/std": 16.109525680541992,
"step": 544
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 505.0,
"completions/mean_length": 311.9375,
"completions/mean_terminated_length": 221.0,
"completions/min_length": 69.0,
"completions/min_terminated_length": 69.0,
"epoch": 0.436,
"grad_norm": 20.555767059326172,
"kl": 3.757080078125,
"learning_rate": 1e-06,
"loss": 0.5049,
"num_tokens": 7307534.0,
"reward": -0.9964599609375,
"reward_std": 8.262689590454102,
"rewards/rm_reward_func/mean": -0.9964599609375,
"rewards/rm_reward_func/std": 15.29664421081543,
"step": 545
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 466.0,
"completions/mean_length": 331.0625,
"completions/mean_terminated_length": 236.2857208251953,
"completions/min_length": 33.0,
"completions/min_terminated_length": 33.0,
"epoch": 0.4368,
"grad_norm": 19.920595169067383,
"kl": 3.9521484375,
"learning_rate": 1e-06,
"loss": 0.0243,
"num_tokens": 7327104.0,
"reward": -12.331245422363281,
"reward_std": 5.995415687561035,
"rewards/rm_reward_func/mean": -12.331245422363281,
"rewards/rm_reward_func/std": 11.88155746459961,
"step": 546
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.5625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 483.0,
"completions/mean_length": 444.3125,
"completions/mean_terminated_length": 357.2857360839844,
"completions/min_length": 68.0,
"completions/min_terminated_length": 68.0,
"epoch": 0.4376,
"grad_norm": 21.573915481567383,
"kl": 4.890625,
"learning_rate": 1e-06,
"loss": 0.3193,
"num_tokens": 7344522.0,
"reward": -5.7215576171875,
"reward_std": 17.706233978271484,
"rewards/rm_reward_func/mean": -5.7215576171875,
"rewards/rm_reward_func/std": 20.520912170410156,
"step": 547
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.5,
"completions/max_length": 512.0,
"completions/max_terminated_length": 460.0,
"completions/mean_length": 403.375,
"completions/mean_terminated_length": 294.75,
"completions/min_length": 76.0,
"completions/min_terminated_length": 76.0,
"epoch": 0.4384,
"grad_norm": 18.524627685546875,
"kl": 5.6640625,
"learning_rate": 1e-06,
"loss": 0.368,
"num_tokens": 7363686.0,
"reward": -8.279296875,
"reward_std": 14.83357048034668,
"rewards/rm_reward_func/mean": -8.279296875,
"rewards/rm_reward_func/std": 16.11411476135254,
"step": 548
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 485.0,
"completions/mean_length": 251.96875,
"completions/mean_terminated_length": 214.82144165039062,
"completions/min_length": 70.0,
"completions/min_terminated_length": 70.0,
"epoch": 0.4392,
"grad_norm": 17.9769344329834,
"kl": 3.47265625,
"learning_rate": 1e-06,
"loss": 0.5011,
"num_tokens": 7374869.0,
"reward": -9.139419555664062,
"reward_std": 6.864280700683594,
"rewards/rm_reward_func/mean": -9.139419555664062,
"rewards/rm_reward_func/std": 13.617587089538574,
"step": 549
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 391.0,
"completions/mean_length": 335.46875,
"completions/mean_terminated_length": 243.0,
"completions/min_length": 74.0,
"completions/min_terminated_length": 74.0,
"epoch": 0.44,
"grad_norm": 12.295485496520996,
"kl": 6.3740234375,
"learning_rate": 1e-06,
"loss": 0.5446,
"num_tokens": 7388164.0,
"reward": -0.984375,
"reward_std": 13.594259262084961,
"rewards/rm_reward_func/mean": -0.984375,
"rewards/rm_reward_func/std": 17.934398651123047,
"step": 550
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.4375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 414.0,
"completions/mean_length": 326.4375,
"completions/mean_terminated_length": 182.11111450195312,
"completions/min_length": 64.0,
"completions/min_terminated_length": 64.0,
"epoch": 0.4408,
"grad_norm": 9.252522468566895,
"kl": 9.421875,
"learning_rate": 1e-06,
"loss": 0.8192,
"num_tokens": 7401826.0,
"reward": -3.5927734375,
"reward_std": 14.570196151733398,
"rewards/rm_reward_func/mean": -3.5927734375,
"rewards/rm_reward_func/std": 18.126157760620117,
"step": 551
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 377.0,
"completions/mean_length": 281.34375,
"completions/mean_terminated_length": 160.52381896972656,
"completions/min_length": 69.0,
"completions/min_terminated_length": 69.0,
"epoch": 0.4416,
"grad_norm": 8.966374397277832,
"kl": 9.5185546875,
"learning_rate": 1e-06,
"loss": 0.6778,
"num_tokens": 7414197.0,
"reward": -1.685546875,
"reward_std": 9.854778289794922,
"rewards/rm_reward_func/mean": -1.685546875,
"rewards/rm_reward_func/std": 13.30488395690918,
"step": 552
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.5625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 398.0,
"completions/mean_length": 359.75,
"completions/mean_terminated_length": 164.0,
"completions/min_length": 75.0,
"completions/min_terminated_length": 75.0,
"epoch": 0.4424,
"grad_norm": 21.584428787231445,
"kl": 13.09375,
"learning_rate": 1e-06,
"loss": 0.9478,
"num_tokens": 7433813.0,
"reward": -11.2989501953125,
"reward_std": 9.919830322265625,
"rewards/rm_reward_func/mean": -11.2989501953125,
"rewards/rm_reward_func/std": 11.97014045715332,
"step": 553
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 466.0,
"completions/mean_length": 406.59375,
"completions/mean_terminated_length": 230.9166717529297,
"completions/min_length": 48.0,
"completions/min_terminated_length": 48.0,
"epoch": 0.4432,
"grad_norm": 63.04143524169922,
"kl": 14.703125,
"learning_rate": 1e-06,
"loss": 0.7666,
"num_tokens": 7449552.0,
"reward": -16.6484375,
"reward_std": 8.551742553710938,
"rewards/rm_reward_func/mean": -16.6484375,
"rewards/rm_reward_func/std": 16.584556579589844,
"step": 554
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.46875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 380.5625,
"completions/mean_terminated_length": 264.5882263183594,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.444,
"grad_norm": 592.711669921875,
"kl": 17.265625,
"learning_rate": 1e-06,
"loss": 1.0223,
"num_tokens": 7465730.0,
"reward": -6.50244140625,
"reward_std": 14.185935974121094,
"rewards/rm_reward_func/mean": -6.50244140625,
"rewards/rm_reward_func/std": 18.073198318481445,
"step": 555
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 438.0,
"completions/mean_length": 319.03125,
"completions/mean_terminated_length": 274.5,
"completions/min_length": 103.0,
"completions/min_terminated_length": 103.0,
"epoch": 0.4448,
"grad_norm": 13.58010482788086,
"kl": 7.970703125,
"learning_rate": 1e-06,
"loss": 0.3343,
"num_tokens": 7478107.0,
"reward": -8.147216796875,
"reward_std": 11.794363021850586,
"rewards/rm_reward_func/mean": -8.147216796875,
"rewards/rm_reward_func/std": 15.557478904724121,
"step": 556
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 493.0,
"completions/mean_length": 273.84375,
"completions/mean_terminated_length": 239.82144165039062,
"completions/min_length": 71.0,
"completions/min_terminated_length": 71.0,
"epoch": 0.4456,
"grad_norm": 25.78789520263672,
"kl": 7.234619140625,
"learning_rate": 1e-06,
"loss": 0.3103,
"num_tokens": 7492406.0,
"reward": 1.95947265625,
"reward_std": 10.006596565246582,
"rewards/rm_reward_func/mean": 1.95947265625,
"rewards/rm_reward_func/std": 15.28310489654541,
"step": 557
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 502.0,
"completions/mean_length": 225.6875,
"completions/mean_terminated_length": 172.6666717529297,
"completions/min_length": 65.0,
"completions/min_terminated_length": 65.0,
"epoch": 0.4464,
"grad_norm": 16.13897705078125,
"kl": 12.6171875,
"learning_rate": 1e-06,
"loss": 1.0411,
"num_tokens": 7504660.0,
"reward": 0.7138671875,
"reward_std": 15.415876388549805,
"rewards/rm_reward_func/mean": 0.7138671875,
"rewards/rm_reward_func/std": 21.385021209716797,
"step": 558
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 348.0,
"completions/mean_terminated_length": 249.60000610351562,
"completions/min_length": 72.0,
"completions/min_terminated_length": 72.0,
"epoch": 0.4472,
"grad_norm": 16.69782829284668,
"kl": 9.87890625,
"learning_rate": 1e-06,
"loss": 0.6521,
"num_tokens": 7518204.0,
"reward": 0.142578125,
"reward_std": 15.55780029296875,
"rewards/rm_reward_func/mean": 0.142578125,
"rewards/rm_reward_func/std": 18.24248695373535,
"step": 559
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 486.0,
"completions/mean_length": 212.71875,
"completions/mean_terminated_length": 169.96429443359375,
"completions/min_length": 24.0,
"completions/min_terminated_length": 24.0,
"epoch": 0.448,
"grad_norm": 96.58130645751953,
"kl": 6.24658203125,
"learning_rate": 1e-06,
"loss": 0.5224,
"num_tokens": 7527819.0,
"reward": 1.61529541015625,
"reward_std": 13.240062713623047,
"rewards/rm_reward_func/mean": 1.61529541015625,
"rewards/rm_reward_func/std": 14.038684844970703,
"step": 560
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 475.0,
"completions/mean_length": 239.09375,
"completions/mean_terminated_length": 210.86207580566406,
"completions/min_length": 27.0,
"completions/min_terminated_length": 27.0,
"epoch": 0.4488,
"grad_norm": 44.059059143066406,
"kl": 11.875,
"learning_rate": 1e-06,
"loss": 0.4508,
"num_tokens": 7538014.0,
"reward": -12.23046875,
"reward_std": 13.231216430664062,
"rewards/rm_reward_func/mean": -12.23046875,
"rewards/rm_reward_func/std": 17.87043571472168,
"step": 561
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 441.0,
"completions/mean_length": 162.28125,
"completions/mean_terminated_length": 126.10344696044922,
"completions/min_length": 42.0,
"completions/min_terminated_length": 42.0,
"epoch": 0.4496,
"grad_norm": 23.391313552856445,
"kl": 6.7119140625,
"learning_rate": 1e-06,
"loss": 0.3591,
"num_tokens": 7546287.0,
"reward": -9.3154296875,
"reward_std": 6.772148132324219,
"rewards/rm_reward_func/mean": -9.3154296875,
"rewards/rm_reward_func/std": 11.018516540527344,
"step": 562
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 479.0,
"completions/mean_length": 366.1875,
"completions/mean_terminated_length": 317.5833435058594,
"completions/min_length": 153.0,
"completions/min_terminated_length": 153.0,
"epoch": 0.4504,
"grad_norm": 41.109046936035156,
"kl": 7.4140625,
"learning_rate": 1e-06,
"loss": 0.3541,
"num_tokens": 7560213.0,
"reward": -8.3702392578125,
"reward_std": 10.981157302856445,
"rewards/rm_reward_func/mean": -8.3702392578125,
"rewards/rm_reward_func/std": 13.88586711883545,
"step": 563
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 483.0,
"completions/mean_length": 339.5,
"completions/mean_terminated_length": 291.1999816894531,
"completions/min_length": 23.0,
"completions/min_terminated_length": 23.0,
"epoch": 0.4512,
"grad_norm": 34.89735412597656,
"kl": 4.27685546875,
"learning_rate": 1e-06,
"loss": 0.0249,
"num_tokens": 7574949.0,
"reward": 1.029541015625,
"reward_std": 12.96914291381836,
"rewards/rm_reward_func/mean": 1.029541015625,
"rewards/rm_reward_func/std": 22.506608963012695,
"step": 564
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 475.0,
"completions/mean_length": 272.375,
"completions/mean_terminated_length": 217.07693481445312,
"completions/min_length": 45.0,
"completions/min_terminated_length": 45.0,
"epoch": 0.452,
"grad_norm": 10.21695613861084,
"kl": 5.6484375,
"learning_rate": 1e-06,
"loss": 0.3193,
"num_tokens": 7586177.0,
"reward": -7.305419921875,
"reward_std": 8.39257526397705,
"rewards/rm_reward_func/mean": -7.305419921875,
"rewards/rm_reward_func/std": 10.208649635314941,
"step": 565
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 490.0,
"completions/mean_length": 196.34375,
"completions/mean_terminated_length": 163.6896514892578,
"completions/min_length": 49.0,
"completions/min_terminated_length": 49.0,
"epoch": 0.4528,
"grad_norm": 74.78276062011719,
"kl": 3.7001953125,
"learning_rate": 1e-06,
"loss": -0.0352,
"num_tokens": 7595588.0,
"reward": -7.03125,
"reward_std": 6.5332746505737305,
"rewards/rm_reward_func/mean": -7.03125,
"rewards/rm_reward_func/std": 15.433568000793457,
"step": 566
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 498.0,
"completions/mean_length": 303.4375,
"completions/mean_terminated_length": 289.5333557128906,
"completions/min_length": 42.0,
"completions/min_terminated_length": 42.0,
"epoch": 0.4536,
"grad_norm": 212.21656799316406,
"kl": 2.8095703125,
"learning_rate": 1e-06,
"loss": -0.0082,
"num_tokens": 7607330.0,
"reward": -1.4860076904296875,
"reward_std": 13.193567276000977,
"rewards/rm_reward_func/mean": -1.4860076904296875,
"rewards/rm_reward_func/std": 17.839399337768555,
"step": 567
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 498.0,
"completions/mean_length": 236.78125,
"completions/mean_terminated_length": 173.2692413330078,
"completions/min_length": 20.0,
"completions/min_terminated_length": 20.0,
"epoch": 0.4544,
"grad_norm": 22.839019775390625,
"kl": 4.947509765625,
"learning_rate": 1e-06,
"loss": 0.3036,
"num_tokens": 7619147.0,
"reward": -4.39599609375,
"reward_std": 10.989629745483398,
"rewards/rm_reward_func/mean": -4.39599609375,
"rewards/rm_reward_func/std": 14.866265296936035,
"step": 568
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 261.25,
"completions/mean_terminated_length": 253.16128540039062,
"completions/min_length": 24.0,
"completions/min_terminated_length": 24.0,
"epoch": 0.4552,
"grad_norm": 10.727757453918457,
"kl": 2.3701171875,
"learning_rate": 1e-06,
"loss": -0.1091,
"num_tokens": 7629659.0,
"reward": 2.4486083984375,
"reward_std": 13.246370315551758,
"rewards/rm_reward_func/mean": 2.4486083984375,
"rewards/rm_reward_func/std": 20.726552963256836,
"step": 569
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 495.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 218.90625,
"completions/mean_terminated_length": 218.90625,
"completions/min_length": 39.0,
"completions/min_terminated_length": 39.0,
"epoch": 0.456,
"grad_norm": 12.797094345092773,
"kl": 2.15869140625,
"learning_rate": 1e-06,
"loss": 0.3033,
"num_tokens": 7641048.0,
"reward": 14.328125,
"reward_std": 6.0160746574401855,
"rewards/rm_reward_func/mean": 14.328125,
"rewards/rm_reward_func/std": 12.738174438476562,
"step": 570
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 443.0,
"completions/mean_length": 239.65625,
"completions/mean_terminated_length": 189.22222900390625,
"completions/min_length": 37.0,
"completions/min_terminated_length": 37.0,
"epoch": 0.4568,
"grad_norm": 28.028533935546875,
"kl": 6.15625,
"learning_rate": 1e-06,
"loss": 0.1429,
"num_tokens": 7651101.0,
"reward": -17.66259765625,
"reward_std": 4.737618446350098,
"rewards/rm_reward_func/mean": -17.66259765625,
"rewards/rm_reward_func/std": 7.526523590087891,
"step": 571
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 493.0,
"completions/mean_length": 244.9375,
"completions/mean_terminated_length": 183.3076934814453,
"completions/min_length": 3.0,
"completions/min_terminated_length": 3.0,
"epoch": 0.4576,
"grad_norm": 28.566667556762695,
"kl": 4.779541015625,
"learning_rate": 1e-06,
"loss": 0.31,
"num_tokens": 7661483.0,
"reward": -11.293212890625,
"reward_std": 4.30594539642334,
"rewards/rm_reward_func/mean": -11.293212890625,
"rewards/rm_reward_func/std": 9.476552963256836,
"step": 572
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 502.0,
"completions/mean_length": 372.5,
"completions/mean_terminated_length": 299.4285888671875,
"completions/min_length": 44.0,
"completions/min_terminated_length": 44.0,
"epoch": 0.4584,
"grad_norm": 9.447068214416504,
"kl": 3.060302734375,
"learning_rate": 1e-06,
"loss": 0.061,
"num_tokens": 7679483.0,
"reward": -9.2685546875,
"reward_std": 7.054027080535889,
"rewards/rm_reward_func/mean": -9.2685546875,
"rewards/rm_reward_func/std": 12.366097450256348,
"step": 573
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 511.0,
"completions/mean_length": 337.125,
"completions/mean_terminated_length": 304.7407531738281,
"completions/min_length": 62.0,
"completions/min_terminated_length": 62.0,
"epoch": 0.4592,
"grad_norm": 11.237780570983887,
"kl": 4.16357421875,
"learning_rate": 1e-06,
"loss": 0.166,
"num_tokens": 7692759.0,
"reward": -8.243682861328125,
"reward_std": 9.03434944152832,
"rewards/rm_reward_func/mean": -8.243682861328125,
"rewards/rm_reward_func/std": 10.684294700622559,
"step": 574
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 259.28125,
"completions/mean_terminated_length": 160.3913116455078,
"completions/min_length": 3.0,
"completions/min_terminated_length": 3.0,
"epoch": 0.46,
"grad_norm": 26.71776008605957,
"kl": 2.917724609375,
"learning_rate": 1e-06,
"loss": 0.0122,
"num_tokens": 7704504.0,
"reward": -0.0777587890625,
"reward_std": 7.303743839263916,
"rewards/rm_reward_func/mean": -0.0777587890625,
"rewards/rm_reward_func/std": 13.445627212524414,
"step": 575
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 478.0,
"completions/mean_length": 329.5625,
"completions/mean_terminated_length": 278.47998046875,
"completions/min_length": 36.0,
"completions/min_terminated_length": 36.0,
"epoch": 0.4608,
"grad_norm": 29.287355422973633,
"kl": 3.5146484375,
"learning_rate": 1e-06,
"loss": 0.1701,
"num_tokens": 7718426.0,
"reward": -9.46527099609375,
"reward_std": 5.940766334533691,
"rewards/rm_reward_func/mean": -9.46527099609375,
"rewards/rm_reward_func/std": 8.79275131225586,
"step": 576
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 498.0,
"completions/mean_length": 271.375,
"completions/mean_terminated_length": 226.8148193359375,
"completions/min_length": 51.0,
"completions/min_terminated_length": 51.0,
"epoch": 0.4616,
"grad_norm": 31.825645446777344,
"kl": 1.86328125,
"learning_rate": 1e-06,
"loss": -0.0067,
"num_tokens": 7731526.0,
"reward": 7.0263671875,
"reward_std": 8.148383140563965,
"rewards/rm_reward_func/mean": 7.0263671875,
"rewards/rm_reward_func/std": 15.413996696472168,
"step": 577
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 470.0,
"completions/mean_length": 321.90625,
"completions/mean_terminated_length": 302.2413635253906,
"completions/min_length": 94.0,
"completions/min_terminated_length": 94.0,
"epoch": 0.4624,
"grad_norm": 9.535029411315918,
"kl": 3.79296875,
"learning_rate": 1e-06,
"loss": 0.0741,
"num_tokens": 7746387.0,
"reward": -4.50048828125,
"reward_std": 10.548614501953125,
"rewards/rm_reward_func/mean": -4.50048828125,
"rewards/rm_reward_func/std": 18.594907760620117,
"step": 578
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 345.0,
"completions/mean_length": 201.5,
"completions/mean_terminated_length": 98.0,
"completions/min_length": 22.0,
"completions/min_terminated_length": 22.0,
"epoch": 0.4632,
"grad_norm": 8.471213340759277,
"kl": 2.734375,
"learning_rate": 1e-06,
"loss": -0.1795,
"num_tokens": 7755707.0,
"reward": -7.9576416015625,
"reward_std": 9.555851936340332,
"rewards/rm_reward_func/mean": -7.9576416015625,
"rewards/rm_reward_func/std": 11.002127647399902,
"step": 579
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 479.0,
"completions/mean_length": 220.96875,
"completions/mean_terminated_length": 190.86207580566406,
"completions/min_length": 63.0,
"completions/min_terminated_length": 63.0,
"epoch": 0.464,
"grad_norm": 8.996735572814941,
"kl": 1.244140625,
"learning_rate": 1e-06,
"loss": -0.0699,
"num_tokens": 7765434.0,
"reward": 3.6036376953125,
"reward_std": 8.55752944946289,
"rewards/rm_reward_func/mean": 3.6036376953125,
"rewards/rm_reward_func/std": 18.499204635620117,
"step": 580
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 487.0,
"completions/mean_length": 346.0,
"completions/mean_terminated_length": 281.0434875488281,
"completions/min_length": 18.0,
"completions/min_terminated_length": 18.0,
"epoch": 0.4648,
"grad_norm": 40.83759689331055,
"kl": 0.305419921875,
"learning_rate": 1e-06,
"loss": 0.1,
"num_tokens": 7778234.0,
"reward": 4.37225341796875,
"reward_std": 7.295747756958008,
"rewards/rm_reward_func/mean": 4.37225341796875,
"rewards/rm_reward_func/std": 8.913907051086426,
"step": 581
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 456.0,
"completions/mean_length": 305.96875,
"completions/mean_terminated_length": 237.2916717529297,
"completions/min_length": 21.0,
"completions/min_terminated_length": 21.0,
"epoch": 0.4656,
"grad_norm": 37.06733703613281,
"kl": 2.0625,
"learning_rate": 1e-06,
"loss": -0.0452,
"num_tokens": 7791017.0,
"reward": -5.02978515625,
"reward_std": 5.757181167602539,
"rewards/rm_reward_func/mean": -5.02978515625,
"rewards/rm_reward_func/std": 9.750814437866211,
"step": 582
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 491.0,
"completions/mean_length": 238.90625,
"completions/mean_terminated_length": 199.8928680419922,
"completions/min_length": 35.0,
"completions/min_terminated_length": 35.0,
"epoch": 0.4664,
"grad_norm": 34.94654846191406,
"kl": 1.6552734375,
"learning_rate": 1e-06,
"loss": -0.1114,
"num_tokens": 7803718.0,
"reward": 1.8824462890625,
"reward_std": 10.042098999023438,
"rewards/rm_reward_func/mean": 1.8824462890625,
"rewards/rm_reward_func/std": 14.9249906539917,
"step": 583
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 494.0,
"completions/mean_length": 289.09375,
"completions/mean_terminated_length": 257.25,
"completions/min_length": 70.0,
"completions/min_terminated_length": 70.0,
"epoch": 0.4672,
"grad_norm": 4.869449138641357,
"kl": 0.384765625,
"learning_rate": 1e-06,
"loss": -0.0789,
"num_tokens": 7815969.0,
"reward": 8.9342041015625,
"reward_std": 9.483150482177734,
"rewards/rm_reward_func/mean": 8.9342041015625,
"rewards/rm_reward_func/std": 18.35944938659668,
"step": 584
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.65625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 462.03125,
"completions/mean_terminated_length": 366.6363830566406,
"completions/min_length": 149.0,
"completions/min_terminated_length": 149.0,
"epoch": 0.468,
"grad_norm": 5.1822285652160645,
"kl": 0.37060546875,
"learning_rate": 1e-06,
"loss": -0.0113,
"num_tokens": 7833826.0,
"reward": 7.9248046875,
"reward_std": 7.081142425537109,
"rewards/rm_reward_func/mean": 7.9248046875,
"rewards/rm_reward_func/std": 15.302535057067871,
"step": 585
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 304.0,
"completions/mean_length": 131.4375,
"completions/mean_terminated_length": 119.16128540039062,
"completions/min_length": 27.0,
"completions/min_terminated_length": 27.0,
"epoch": 0.4688,
"grad_norm": 18.64679718017578,
"kl": 3.083984375,
"learning_rate": 1e-06,
"loss": 0.2691,
"num_tokens": 7841144.0,
"reward": 5.0889892578125,
"reward_std": 6.059321880340576,
"rewards/rm_reward_func/mean": 5.0889892578125,
"rewards/rm_reward_func/std": 9.212634086608887,
"step": 586
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 497.0,
"completions/mean_length": 335.84375,
"completions/mean_terminated_length": 310.6785888671875,
"completions/min_length": 27.0,
"completions/min_terminated_length": 27.0,
"epoch": 0.4696,
"grad_norm": 8.921401023864746,
"kl": 2.075439453125,
"learning_rate": 1e-06,
"loss": -0.1186,
"num_tokens": 7853947.0,
"reward": -0.1695556640625,
"reward_std": 9.398602485656738,
"rewards/rm_reward_func/mean": -0.1695556640625,
"rewards/rm_reward_func/std": 15.199664115905762,
"step": 587
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 491.0,
"completions/mean_length": 301.5,
"completions/mean_terminated_length": 271.4285888671875,
"completions/min_length": 30.0,
"completions/min_terminated_length": 30.0,
"epoch": 0.4704,
"grad_norm": 49.62898635864258,
"kl": 0.86962890625,
"learning_rate": 1e-06,
"loss": -0.058,
"num_tokens": 7865547.0,
"reward": -1.8533935546875,
"reward_std": 7.218017578125,
"rewards/rm_reward_func/mean": -1.8533935546875,
"rewards/rm_reward_func/std": 9.405008316040039,
"step": 588
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 503.0,
"completions/max_terminated_length": 503.0,
"completions/mean_length": 333.375,
"completions/mean_terminated_length": 333.375,
"completions/min_length": 171.0,
"completions/min_terminated_length": 171.0,
"epoch": 0.4712,
"grad_norm": 15.543201446533203,
"kl": 0.32958984375,
"learning_rate": 1e-06,
"loss": -0.0359,
"num_tokens": 7879767.0,
"reward": -5.37200927734375,
"reward_std": 3.4672133922576904,
"rewards/rm_reward_func/mean": -5.37200927734375,
"rewards/rm_reward_func/std": 7.8242692947387695,
"step": 589
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 377.84375,
"completions/mean_terminated_length": 340.2799987792969,
"completions/min_length": 94.0,
"completions/min_terminated_length": 94.0,
"epoch": 0.472,
"grad_norm": 7.237336158752441,
"kl": 0.4638671875,
"learning_rate": 1e-06,
"loss": -0.0329,
"num_tokens": 7894162.0,
"reward": 6.933837890625,
"reward_std": 7.349064826965332,
"rewards/rm_reward_func/mean": 6.933837890625,
"rewards/rm_reward_func/std": 12.662103652954102,
"step": 590
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 270.09375,
"completions/mean_terminated_length": 245.0689697265625,
"completions/min_length": 74.0,
"completions/min_terminated_length": 74.0,
"epoch": 0.4728,
"grad_norm": 9.466930389404297,
"kl": 1.3511962890625,
"learning_rate": 1e-06,
"loss": 0.1344,
"num_tokens": 7905197.0,
"reward": -5.54095458984375,
"reward_std": 5.511334419250488,
"rewards/rm_reward_func/mean": -5.54095458984375,
"rewards/rm_reward_func/std": 9.380617141723633,
"step": 591
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 511.0,
"completions/mean_length": 187.5625,
"completions/mean_terminated_length": 127.48148345947266,
"completions/min_length": 30.0,
"completions/min_terminated_length": 30.0,
"epoch": 0.4736,
"grad_norm": 10.78067684173584,
"kl": 0.6640625,
"learning_rate": 1e-06,
"loss": -0.0794,
"num_tokens": 7913271.0,
"reward": 4.76513671875,
"reward_std": 7.13428258895874,
"rewards/rm_reward_func/mean": 4.76513671875,
"rewards/rm_reward_func/std": 9.546843528747559,
"step": 592
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 500.0,
"completions/mean_length": 297.8125,
"completions/mean_terminated_length": 214.0,
"completions/min_length": 69.0,
"completions/min_terminated_length": 69.0,
"epoch": 0.4744,
"grad_norm": 5.899567604064941,
"kl": 0.46875,
"learning_rate": 1e-06,
"loss": -0.1007,
"num_tokens": 7926073.0,
"reward": 1.1571044921875,
"reward_std": 7.337251663208008,
"rewards/rm_reward_func/mean": 1.1571044921875,
"rewards/rm_reward_func/std": 10.56173038482666,
"step": 593
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.4375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 446.40625,
"completions/mean_terminated_length": 395.3888854980469,
"completions/min_length": 261.0,
"completions/min_terminated_length": 261.0,
"epoch": 0.4752,
"grad_norm": 41.35441589355469,
"kl": 2.393310546875,
"learning_rate": 1e-06,
"loss": 0.12,
"num_tokens": 7942310.0,
"reward": -0.016571044921875,
"reward_std": 9.10225772857666,
"rewards/rm_reward_func/mean": -0.016571044921875,
"rewards/rm_reward_func/std": 14.169840812683105,
"step": 594
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 468.0,
"completions/mean_length": 234.15625,
"completions/mean_terminated_length": 215.6333465576172,
"completions/min_length": 54.0,
"completions/min_terminated_length": 54.0,
"epoch": 0.476,
"grad_norm": 13.656573295593262,
"kl": 6.201904296875,
"learning_rate": 1e-06,
"loss": 0.7573,
"num_tokens": 7954699.0,
"reward": 6.20806884765625,
"reward_std": 9.577255249023438,
"rewards/rm_reward_func/mean": 6.20806884765625,
"rewards/rm_reward_func/std": 14.503737449645996,
"step": 595
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.5,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 455.46875,
"completions/mean_terminated_length": 398.9375,
"completions/min_length": 202.0,
"completions/min_terminated_length": 202.0,
"epoch": 0.4768,
"grad_norm": 5.957422256469727,
"kl": 2.294677734375,
"learning_rate": 1e-06,
"loss": 0.071,
"num_tokens": 7973066.0,
"reward": -7.954833984375,
"reward_std": 6.558610916137695,
"rewards/rm_reward_func/mean": -7.954833984375,
"rewards/rm_reward_func/std": 9.106958389282227,
"step": 596
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 506.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 269.96875,
"completions/mean_terminated_length": 269.96875,
"completions/min_length": 65.0,
"completions/min_terminated_length": 65.0,
"epoch": 0.4776,
"grad_norm": 8.821357727050781,
"kl": 0.6884765625,
"learning_rate": 1e-06,
"loss": 0.0222,
"num_tokens": 7984641.0,
"reward": 6.18316650390625,
"reward_std": 5.323254585266113,
"rewards/rm_reward_func/mean": 6.18316650390625,
"rewards/rm_reward_func/std": 14.378352165222168,
"step": 597
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 501.0,
"completions/mean_length": 231.8125,
"completions/mean_terminated_length": 222.77418518066406,
"completions/min_length": 66.0,
"completions/min_terminated_length": 66.0,
"epoch": 0.4784,
"grad_norm": 34.12923812866211,
"kl": 7.343505859375,
"learning_rate": 1e-06,
"loss": 0.3296,
"num_tokens": 7995091.0,
"reward": 2.2025146484375,
"reward_std": 7.2968525886535645,
"rewards/rm_reward_func/mean": 2.2025146484375,
"rewards/rm_reward_func/std": 13.586483001708984,
"step": 598
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 502.0,
"completions/mean_length": 302.03125,
"completions/mean_terminated_length": 280.3103332519531,
"completions/min_length": 73.0,
"completions/min_terminated_length": 73.0,
"epoch": 0.4792,
"grad_norm": 6.906167507171631,
"kl": 0.790283203125,
"learning_rate": 1e-06,
"loss": -0.1121,
"num_tokens": 8007900.0,
"reward": 4.49267578125,
"reward_std": 6.870022296905518,
"rewards/rm_reward_func/mean": 4.49267578125,
"rewards/rm_reward_func/std": 9.993231773376465,
"step": 599
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 474.0,
"completions/mean_length": 248.53125,
"completions/mean_terminated_length": 210.8928680419922,
"completions/min_length": 17.0,
"completions/min_terminated_length": 17.0,
"epoch": 0.48,
"grad_norm": 31.199399948120117,
"kl": 5.575927734375,
"learning_rate": 1e-06,
"loss": 0.4506,
"num_tokens": 8022373.0,
"reward": -0.75750732421875,
"reward_std": 8.437265396118164,
"rewards/rm_reward_func/mean": -0.75750732421875,
"rewards/rm_reward_func/std": 14.56766128540039,
"step": 600
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 446.0,
"completions/mean_length": 286.75,
"completions/mean_terminated_length": 234.7692413330078,
"completions/min_length": 57.0,
"completions/min_terminated_length": 57.0,
"epoch": 0.4808,
"grad_norm": 61.780311584472656,
"kl": 6.131103515625,
"learning_rate": 1e-06,
"loss": 0.1509,
"num_tokens": 8034205.0,
"reward": 8.10595703125,
"reward_std": 10.480352401733398,
"rewards/rm_reward_func/mean": 8.10595703125,
"rewards/rm_reward_func/std": 14.536432266235352,
"step": 601
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 477.0,
"completions/mean_length": 346.59375,
"completions/mean_terminated_length": 259.952392578125,
"completions/min_length": 75.0,
"completions/min_terminated_length": 75.0,
"epoch": 0.4816,
"grad_norm": 213.77684020996094,
"kl": 21.95849609375,
"learning_rate": 1e-06,
"loss": 1.2326,
"num_tokens": 8048464.0,
"reward": -6.7283935546875,
"reward_std": 6.977231979370117,
"rewards/rm_reward_func/mean": -6.7283935546875,
"rewards/rm_reward_func/std": 14.559057235717773,
"step": 602
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 491.0,
"completions/mean_length": 360.71875,
"completions/mean_terminated_length": 318.3599853515625,
"completions/min_length": 78.0,
"completions/min_terminated_length": 78.0,
"epoch": 0.4824,
"grad_norm": 13.767047882080078,
"kl": 8.62158203125,
"learning_rate": 1e-06,
"loss": 0.4078,
"num_tokens": 8063103.0,
"reward": -1.645751953125,
"reward_std": 12.11515998840332,
"rewards/rm_reward_func/mean": -1.645751953125,
"rewards/rm_reward_func/std": 15.272088050842285,
"step": 603
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 481.0,
"completions/mean_length": 322.0,
"completions/mean_terminated_length": 278.15386962890625,
"completions/min_length": 46.0,
"completions/min_terminated_length": 46.0,
"epoch": 0.4832,
"grad_norm": 24.88362693786621,
"kl": 8.8076171875,
"learning_rate": 1e-06,
"loss": 0.2726,
"num_tokens": 8075391.0,
"reward": -6.1571044921875,
"reward_std": 18.25802230834961,
"rewards/rm_reward_func/mean": -6.1571044921875,
"rewards/rm_reward_func/std": 19.966968536376953,
"step": 604
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 453.0,
"completions/mean_length": 218.71875,
"completions/mean_terminated_length": 176.82144165039062,
"completions/min_length": 58.0,
"completions/min_terminated_length": 58.0,
"epoch": 0.484,
"grad_norm": 11.096504211425781,
"kl": 3.28759765625,
"learning_rate": 1e-06,
"loss": 0.169,
"num_tokens": 8086822.0,
"reward": 6.87744140625,
"reward_std": 7.8228960037231445,
"rewards/rm_reward_func/mean": 6.87744140625,
"rewards/rm_reward_func/std": 12.922216415405273,
"step": 605
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 420.03125,
"completions/mean_terminated_length": 398.8077087402344,
"completions/min_length": 201.0,
"completions/min_terminated_length": 201.0,
"epoch": 0.4848,
"grad_norm": 10.59483528137207,
"kl": 3.473876953125,
"learning_rate": 1e-06,
"loss": 0.0768,
"num_tokens": 8106231.0,
"reward": 15.840576171875,
"reward_std": 15.33732795715332,
"rewards/rm_reward_func/mean": 15.840576171875,
"rewards/rm_reward_func/std": 20.74690818786621,
"step": 606
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 488.0,
"completions/mean_length": 429.15625,
"completions/mean_terminated_length": 385.76190185546875,
"completions/min_length": 167.0,
"completions/min_terminated_length": 167.0,
"epoch": 0.4856,
"grad_norm": 9.831181526184082,
"kl": 4.218505859375,
"learning_rate": 1e-06,
"loss": 0.1415,
"num_tokens": 8124388.0,
"reward": -1.52099609375,
"reward_std": 13.66183853149414,
"rewards/rm_reward_func/mean": -1.52099609375,
"rewards/rm_reward_func/std": 18.234188079833984,
"step": 607
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 358.75,
"completions/mean_terminated_length": 323.3846130371094,
"completions/min_length": 14.0,
"completions/min_terminated_length": 14.0,
"epoch": 0.4864,
"grad_norm": 10.595935821533203,
"kl": 2.94921875,
"learning_rate": 1e-06,
"loss": -0.0267,
"num_tokens": 8138108.0,
"reward": 5.33203125,
"reward_std": 11.863018035888672,
"rewards/rm_reward_func/mean": 5.33203125,
"rewards/rm_reward_func/std": 24.679624557495117,
"step": 608
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 460.0,
"completions/mean_length": 217.8125,
"completions/mean_terminated_length": 208.32257080078125,
"completions/min_length": 53.0,
"completions/min_terminated_length": 53.0,
"epoch": 0.4872,
"grad_norm": 22.514549255371094,
"kl": 5.7451171875,
"learning_rate": 1e-06,
"loss": 0.3125,
"num_tokens": 8150726.0,
"reward": 0.619384765625,
"reward_std": 10.594083786010742,
"rewards/rm_reward_func/mean": 0.619384765625,
"rewards/rm_reward_func/std": 15.584603309631348,
"step": 609
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 392.0,
"completions/mean_length": 236.28125,
"completions/mean_terminated_length": 207.7586212158203,
"completions/min_length": 56.0,
"completions/min_terminated_length": 56.0,
"epoch": 0.488,
"grad_norm": 25.03614616394043,
"kl": 8.80859375,
"learning_rate": 1e-06,
"loss": 0.7636,
"num_tokens": 8162679.0,
"reward": 4.9461517333984375,
"reward_std": 10.481914520263672,
"rewards/rm_reward_func/mean": 4.9461517333984375,
"rewards/rm_reward_func/std": 14.789458274841309,
"step": 610
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 405.28125,
"completions/mean_terminated_length": 363.5217590332031,
"completions/min_length": 72.0,
"completions/min_terminated_length": 72.0,
"epoch": 0.4888,
"grad_norm": 9.863654136657715,
"kl": 4.548828125,
"learning_rate": 1e-06,
"loss": 0.1808,
"num_tokens": 8177880.0,
"reward": 1.1624755859375,
"reward_std": 12.410948753356934,
"rewards/rm_reward_func/mean": 1.1624755859375,
"rewards/rm_reward_func/std": 17.47865867614746,
"step": 611
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 382.3125,
"completions/mean_terminated_length": 358.2962951660156,
"completions/min_length": 162.0,
"completions/min_terminated_length": 162.0,
"epoch": 0.4896,
"grad_norm": 16.073301315307617,
"kl": 3.01123046875,
"learning_rate": 1e-06,
"loss": 0.0967,
"num_tokens": 8192650.0,
"reward": 0.911651611328125,
"reward_std": 8.299646377563477,
"rewards/rm_reward_func/mean": 0.911651611328125,
"rewards/rm_reward_func/std": 14.285723686218262,
"step": 612
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 450.0,
"completions/mean_length": 341.65625,
"completions/mean_terminated_length": 275.0,
"completions/min_length": 67.0,
"completions/min_terminated_length": 67.0,
"epoch": 0.4904,
"grad_norm": 37.6219596862793,
"kl": 9.759765625,
"learning_rate": 1e-06,
"loss": 0.5727,
"num_tokens": 8205711.0,
"reward": -1.9541015625,
"reward_std": 9.328568458557129,
"rewards/rm_reward_func/mean": -1.9541015625,
"rewards/rm_reward_func/std": 21.1943302154541,
"step": 613
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 474.0,
"completions/mean_length": 267.03125,
"completions/mean_terminated_length": 185.375,
"completions/min_length": 32.0,
"completions/min_terminated_length": 32.0,
"epoch": 0.4912,
"grad_norm": 153.5972900390625,
"kl": 7.2421875,
"learning_rate": 1e-06,
"loss": 0.3504,
"num_tokens": 8218824.0,
"reward": -9.59063720703125,
"reward_std": 9.781373023986816,
"rewards/rm_reward_func/mean": -9.59063720703125,
"rewards/rm_reward_func/std": 11.414146423339844,
"step": 614
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 492.0,
"completions/mean_length": 188.53125,
"completions/mean_terminated_length": 166.9666748046875,
"completions/min_length": 12.0,
"completions/min_terminated_length": 12.0,
"epoch": 0.492,
"grad_norm": 28.16669464111328,
"kl": 5.5712890625,
"learning_rate": 1e-06,
"loss": 0.1814,
"num_tokens": 8228137.0,
"reward": -3.1532211303710938,
"reward_std": 12.097888946533203,
"rewards/rm_reward_func/mean": -3.1532211303710938,
"rewards/rm_reward_func/std": 20.61545181274414,
"step": 615
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 490.0,
"completions/mean_length": 313.53125,
"completions/mean_terminated_length": 235.86956787109375,
"completions/min_length": 43.0,
"completions/min_terminated_length": 43.0,
"epoch": 0.4928,
"grad_norm": 40.259830474853516,
"kl": 10.94140625,
"learning_rate": 1e-06,
"loss": 0.5348,
"num_tokens": 8241418.0,
"reward": -5.6474609375,
"reward_std": 11.151725769042969,
"rewards/rm_reward_func/mean": -5.6474609375,
"rewards/rm_reward_func/std": 13.004457473754883,
"step": 616
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 500.0,
"completions/mean_length": 265.28125,
"completions/mean_terminated_length": 230.0357208251953,
"completions/min_length": 30.0,
"completions/min_terminated_length": 30.0,
"epoch": 0.4936,
"grad_norm": 9.916903495788574,
"kl": 5.4296875,
"learning_rate": 1e-06,
"loss": 0.1988,
"num_tokens": 8251955.0,
"reward": -0.9298095703125,
"reward_std": 15.462509155273438,
"rewards/rm_reward_func/mean": -0.9298095703125,
"rewards/rm_reward_func/std": 16.872419357299805,
"step": 617
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 471.0,
"completions/mean_length": 204.09375,
"completions/mean_terminated_length": 172.2413787841797,
"completions/min_length": 25.0,
"completions/min_terminated_length": 25.0,
"epoch": 0.4944,
"grad_norm": 63.2779655456543,
"kl": 8.203125,
"learning_rate": 1e-06,
"loss": 0.2805,
"num_tokens": 8264246.0,
"reward": -4.11572265625,
"reward_std": 13.06856918334961,
"rewards/rm_reward_func/mean": -4.11572265625,
"rewards/rm_reward_func/std": 14.47628116607666,
"step": 618
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 448.0,
"completions/mean_length": 346.78125,
"completions/mean_terminated_length": 247.65000915527344,
"completions/min_length": 28.0,
"completions/min_terminated_length": 28.0,
"epoch": 0.4952,
"grad_norm": 44.88002014160156,
"kl": 9.136474609375,
"learning_rate": 1e-06,
"loss": 0.5063,
"num_tokens": 8277079.0,
"reward": -9.4381103515625,
"reward_std": 5.765872955322266,
"rewards/rm_reward_func/mean": -9.4381103515625,
"rewards/rm_reward_func/std": 11.600118637084961,
"step": 619
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 480.0,
"completions/mean_length": 293.15625,
"completions/mean_terminated_length": 270.5172424316406,
"completions/min_length": 34.0,
"completions/min_terminated_length": 34.0,
"epoch": 0.496,
"grad_norm": 37.431556701660156,
"kl": 2.302734375,
"learning_rate": 1e-06,
"loss": 0.0574,
"num_tokens": 8288420.0,
"reward": -2.6103515625,
"reward_std": 13.910895347595215,
"rewards/rm_reward_func/mean": -2.6103515625,
"rewards/rm_reward_func/std": 16.85130500793457,
"step": 620
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 473.0,
"completions/mean_length": 258.84375,
"completions/mean_terminated_length": 222.6785888671875,
"completions/min_length": 97.0,
"completions/min_terminated_length": 97.0,
"epoch": 0.4968,
"grad_norm": 13.83285903930664,
"kl": 7.1640625,
"learning_rate": 1e-06,
"loss": 0.2555,
"num_tokens": 8298487.0,
"reward": -6.125762939453125,
"reward_std": 8.449478149414062,
"rewards/rm_reward_func/mean": -6.125762939453125,
"rewards/rm_reward_func/std": 9.483611106872559,
"step": 621
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 435.0,
"completions/mean_length": 191.71875,
"completions/mean_terminated_length": 170.36666870117188,
"completions/min_length": 21.0,
"completions/min_terminated_length": 21.0,
"epoch": 0.4976,
"grad_norm": 12.294360160827637,
"kl": 5.13671875,
"learning_rate": 1e-06,
"loss": 0.2798,
"num_tokens": 8310150.0,
"reward": 7.74169921875,
"reward_std": 11.26545238494873,
"rewards/rm_reward_func/mean": 7.74169921875,
"rewards/rm_reward_func/std": 15.538458824157715,
"step": 622
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 461.0,
"completions/mean_length": 234.125,
"completions/mean_terminated_length": 170.0,
"completions/min_length": 63.0,
"completions/min_terminated_length": 63.0,
"epoch": 0.4984,
"grad_norm": 7.7052130699157715,
"kl": 2.001953125,
"learning_rate": 1e-06,
"loss": 0.0997,
"num_tokens": 8321794.0,
"reward": 6.1068115234375,
"reward_std": 6.466403961181641,
"rewards/rm_reward_func/mean": 6.1068115234375,
"rewards/rm_reward_func/std": 10.666815757751465,
"step": 623
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 489.0,
"completions/mean_length": 271.15625,
"completions/mean_terminated_length": 226.55555725097656,
"completions/min_length": 51.0,
"completions/min_terminated_length": 51.0,
"epoch": 0.4992,
"grad_norm": 37.59298324584961,
"kl": 6.85546875,
"learning_rate": 1e-06,
"loss": 0.4438,
"num_tokens": 8333623.0,
"reward": -11.0943603515625,
"reward_std": 6.584903717041016,
"rewards/rm_reward_func/mean": -11.0943603515625,
"rewards/rm_reward_func/std": 11.544700622558594,
"step": 624
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 454.0,
"completions/mean_length": 282.40625,
"completions/mean_terminated_length": 205.875,
"completions/min_length": 8.0,
"completions/min_terminated_length": 8.0,
"epoch": 0.5,
"grad_norm": 50.945106506347656,
"kl": 7.80419921875,
"learning_rate": 1e-06,
"loss": 0.1727,
"num_tokens": 8346988.0,
"reward": -10.51025390625,
"reward_std": 4.640988349914551,
"rewards/rm_reward_func/mean": -10.51025390625,
"rewards/rm_reward_func/std": 19.07770538330078,
"step": 625
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 276.5625,
"completions/mean_terminated_length": 260.8666687011719,
"completions/min_length": 42.0,
"completions/min_terminated_length": 42.0,
"epoch": 0.5008,
"grad_norm": 8.402812957763672,
"kl": 3.9267578125,
"learning_rate": 1e-06,
"loss": 0.0878,
"num_tokens": 8358310.0,
"reward": 10.74951171875,
"reward_std": 11.159141540527344,
"rewards/rm_reward_func/mean": 10.74951171875,
"rewards/rm_reward_func/std": 24.454740524291992,
"step": 626
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 487.0,
"completions/mean_length": 314.59375,
"completions/mean_terminated_length": 294.17242431640625,
"completions/min_length": 66.0,
"completions/min_terminated_length": 66.0,
"epoch": 0.5016,
"grad_norm": 7.214862823486328,
"kl": 3.716796875,
"learning_rate": 1e-06,
"loss": 0.1032,
"num_tokens": 8371249.0,
"reward": -1.5455322265625,
"reward_std": 13.805252075195312,
"rewards/rm_reward_func/mean": -1.5455322265625,
"rewards/rm_reward_func/std": 15.483540534973145,
"step": 627
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 311.90625,
"completions/mean_terminated_length": 255.87998962402344,
"completions/min_length": 47.0,
"completions/min_terminated_length": 47.0,
"epoch": 0.5024,
"grad_norm": 5.6593217849731445,
"kl": 2.5107421875,
"learning_rate": 1e-06,
"loss": -0.0017,
"num_tokens": 8383638.0,
"reward": 5.8603515625,
"reward_std": 11.585649490356445,
"rewards/rm_reward_func/mean": 5.8603515625,
"rewards/rm_reward_func/std": 14.266550064086914,
"step": 628
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 500.0,
"completions/mean_length": 272.90625,
"completions/mean_terminated_length": 238.75001525878906,
"completions/min_length": 65.0,
"completions/min_terminated_length": 65.0,
"epoch": 0.5032,
"grad_norm": 11.517411231994629,
"kl": 6.453125,
"learning_rate": 1e-06,
"loss": 0.2772,
"num_tokens": 8394563.0,
"reward": -4.296142578125,
"reward_std": 15.2503023147583,
"rewards/rm_reward_func/mean": -4.296142578125,
"rewards/rm_reward_func/std": 15.96473217010498,
"step": 629
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 497.0,
"completions/mean_length": 310.53125,
"completions/mean_terminated_length": 281.75,
"completions/min_length": 43.0,
"completions/min_terminated_length": 43.0,
"epoch": 0.504,
"grad_norm": 6.986127853393555,
"kl": 1.2666015625,
"learning_rate": 1e-06,
"loss": -0.0682,
"num_tokens": 8410244.0,
"reward": 1.4921875,
"reward_std": 4.761030673980713,
"rewards/rm_reward_func/mean": 1.4921875,
"rewards/rm_reward_func/std": 14.25729751586914,
"step": 630
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 494.0,
"completions/mean_length": 302.84375,
"completions/mean_terminated_length": 288.9000244140625,
"completions/min_length": 68.0,
"completions/min_terminated_length": 68.0,
"epoch": 0.5048,
"grad_norm": 7.896935939788818,
"kl": 4.4833984375,
"learning_rate": 1e-06,
"loss": 0.0858,
"num_tokens": 8424207.0,
"reward": 5.5654296875,
"reward_std": 8.17580509185791,
"rewards/rm_reward_func/mean": 5.5654296875,
"rewards/rm_reward_func/std": 20.969762802124023,
"step": 631
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 271.1875,
"completions/mean_terminated_length": 246.27586364746094,
"completions/min_length": 42.0,
"completions/min_terminated_length": 42.0,
"epoch": 0.5056,
"grad_norm": 10.68520450592041,
"kl": 1.8544921875,
"learning_rate": 1e-06,
"loss": 0.1494,
"num_tokens": 8436013.0,
"reward": 1.7723541259765625,
"reward_std": 9.819549560546875,
"rewards/rm_reward_func/mean": 1.7723541259765625,
"rewards/rm_reward_func/std": 15.363565444946289,
"step": 632
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 500.0,
"completions/mean_length": 327.875,
"completions/mean_terminated_length": 315.6000061035156,
"completions/min_length": 115.0,
"completions/min_terminated_length": 115.0,
"epoch": 0.5064,
"grad_norm": 8.58014965057373,
"kl": 1.8515625,
"learning_rate": 1e-06,
"loss": 0.1069,
"num_tokens": 8449793.0,
"reward": 6.67364501953125,
"reward_std": 11.383968353271484,
"rewards/rm_reward_func/mean": 6.67364501953125,
"rewards/rm_reward_func/std": 13.357749938964844,
"step": 633
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 450.0,
"completions/mean_length": 272.25,
"completions/mean_terminated_length": 205.1199951171875,
"completions/min_length": 99.0,
"completions/min_terminated_length": 99.0,
"epoch": 0.5072,
"grad_norm": 9.852457046508789,
"kl": 2.81201171875,
"learning_rate": 1e-06,
"loss": 0.135,
"num_tokens": 8460897.0,
"reward": -4.5697021484375,
"reward_std": 5.767214775085449,
"rewards/rm_reward_func/mean": -4.5697021484375,
"rewards/rm_reward_func/std": 9.804206848144531,
"step": 634
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 511.0,
"completions/mean_length": 277.375,
"completions/mean_terminated_length": 253.10345458984375,
"completions/min_length": 48.0,
"completions/min_terminated_length": 48.0,
"epoch": 0.508,
"grad_norm": 13.91521167755127,
"kl": 4.9462890625,
"learning_rate": 1e-06,
"loss": 0.1343,
"num_tokens": 8472789.0,
"reward": 0.547698974609375,
"reward_std": 12.389558792114258,
"rewards/rm_reward_func/mean": 0.547698974609375,
"rewards/rm_reward_func/std": 21.48679542541504,
"step": 635
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 488.0,
"completions/mean_length": 246.15625,
"completions/mean_terminated_length": 196.92593383789062,
"completions/min_length": 67.0,
"completions/min_terminated_length": 67.0,
"epoch": 0.5088,
"grad_norm": 10.53650188446045,
"kl": 2.0712890625,
"learning_rate": 1e-06,
"loss": 0.1688,
"num_tokens": 8484914.0,
"reward": -1.3759765625,
"reward_std": 2.3449654579162598,
"rewards/rm_reward_func/mean": -1.3759765625,
"rewards/rm_reward_func/std": 12.400309562683105,
"step": 636
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 488.0,
"completions/mean_length": 333.875,
"completions/mean_terminated_length": 264.1739196777344,
"completions/min_length": 50.0,
"completions/min_terminated_length": 50.0,
"epoch": 0.5096,
"grad_norm": 24.877025604248047,
"kl": 6.34228515625,
"learning_rate": 1e-06,
"loss": 0.2245,
"num_tokens": 8500902.0,
"reward": -9.4365234375,
"reward_std": 3.848327159881592,
"rewards/rm_reward_func/mean": -9.4365234375,
"rewards/rm_reward_func/std": 13.026070594787598,
"step": 637
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 410.40625,
"completions/mean_terminated_length": 364.227294921875,
"completions/min_length": 92.0,
"completions/min_terminated_length": 92.0,
"epoch": 0.5104,
"grad_norm": 8.95622730255127,
"kl": 1.86279296875,
"learning_rate": 1e-06,
"loss": -0.0112,
"num_tokens": 8515907.0,
"reward": 1.7091064453125,
"reward_std": 10.488420486450195,
"rewards/rm_reward_func/mean": 1.7091064453125,
"rewards/rm_reward_func/std": 15.33460807800293,
"step": 638
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 507.0,
"completions/mean_length": 274.0625,
"completions/mean_terminated_length": 207.44000244140625,
"completions/min_length": 68.0,
"completions/min_terminated_length": 68.0,
"epoch": 0.5112,
"grad_norm": 13.682821273803711,
"kl": 5.75537109375,
"learning_rate": 1e-06,
"loss": 0.0707,
"num_tokens": 8528813.0,
"reward": -12.8310546875,
"reward_std": 8.058960914611816,
"rewards/rm_reward_func/mean": -12.8310546875,
"rewards/rm_reward_func/std": 10.044072151184082,
"step": 639
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 252.125,
"completions/mean_terminated_length": 215.00001525878906,
"completions/min_length": 67.0,
"completions/min_terminated_length": 67.0,
"epoch": 0.512,
"grad_norm": 10.187383651733398,
"kl": 3.99658203125,
"learning_rate": 1e-06,
"loss": 0.1334,
"num_tokens": 8539441.0,
"reward": 1.024169921875,
"reward_std": 7.485830783843994,
"rewards/rm_reward_func/mean": 1.024169921875,
"rewards/rm_reward_func/std": 11.878993034362793,
"step": 640
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 197.96875,
"completions/mean_terminated_length": 139.8148193359375,
"completions/min_length": 67.0,
"completions/min_terminated_length": 67.0,
"epoch": 0.5128,
"grad_norm": 6.773480415344238,
"kl": 4.6123046875,
"learning_rate": 1e-06,
"loss": 0.2292,
"num_tokens": 8549136.0,
"reward": -1.02978515625,
"reward_std": 5.95222282409668,
"rewards/rm_reward_func/mean": -1.02978515625,
"rewards/rm_reward_func/std": 9.468921661376953,
"step": 641
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 480.0,
"completions/mean_length": 213.53125,
"completions/mean_terminated_length": 170.8928680419922,
"completions/min_length": 67.0,
"completions/min_terminated_length": 67.0,
"epoch": 0.5136,
"grad_norm": 6.013530254364014,
"kl": 0.498046875,
"learning_rate": 1e-06,
"loss": 0.0151,
"num_tokens": 8560257.0,
"reward": 8.07672119140625,
"reward_std": 5.212985038757324,
"rewards/rm_reward_func/mean": 8.07672119140625,
"rewards/rm_reward_func/std": 9.585315704345703,
"step": 642
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 505.0,
"completions/mean_length": 419.125,
"completions/mean_terminated_length": 393.1199951171875,
"completions/min_length": 233.0,
"completions/min_terminated_length": 233.0,
"epoch": 0.5144,
"grad_norm": 6.766341686248779,
"kl": 1.09765625,
"learning_rate": 1e-06,
"loss": 0.0954,
"num_tokens": 8575533.0,
"reward": 14.398681640625,
"reward_std": 10.005803108215332,
"rewards/rm_reward_func/mean": 14.398681640625,
"rewards/rm_reward_func/std": 12.649359703063965,
"step": 643
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 221.21875,
"completions/mean_terminated_length": 167.37037658691406,
"completions/min_length": 67.0,
"completions/min_terminated_length": 67.0,
"epoch": 0.5152,
"grad_norm": 15.727855682373047,
"kl": 0.90576171875,
"learning_rate": 1e-06,
"loss": -0.0352,
"num_tokens": 8587596.0,
"reward": 7.632568359375,
"reward_std": 3.306748867034912,
"rewards/rm_reward_func/mean": 7.632568359375,
"rewards/rm_reward_func/std": 9.895013809204102,
"step": 644
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 465.0,
"completions/mean_length": 252.40625,
"completions/mean_terminated_length": 235.10000610351562,
"completions/min_length": 69.0,
"completions/min_terminated_length": 69.0,
"epoch": 0.516,
"grad_norm": 20.60538673400879,
"kl": 1.96630859375,
"learning_rate": 1e-06,
"loss": -0.0003,
"num_tokens": 8599505.0,
"reward": 5.3416748046875,
"reward_std": 11.507960319519043,
"rewards/rm_reward_func/mean": 5.3416748046875,
"rewards/rm_reward_func/std": 17.1043758392334,
"step": 645
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.46875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 469.0,
"completions/mean_length": 281.15625,
"completions/mean_terminated_length": 77.47058868408203,
"completions/min_length": 31.0,
"completions/min_terminated_length": 31.0,
"epoch": 0.5168,
"grad_norm": 10.928936004638672,
"kl": 1.356201171875,
"learning_rate": 1e-06,
"loss": 0.0367,
"num_tokens": 8610998.0,
"reward": -1.3593597412109375,
"reward_std": 3.8079264163970947,
"rewards/rm_reward_func/mean": -1.3593597412109375,
"rewards/rm_reward_func/std": 8.6522216796875,
"step": 646
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 486.0,
"completions/mean_length": 362.59375,
"completions/mean_terminated_length": 312.79168701171875,
"completions/min_length": 47.0,
"completions/min_terminated_length": 47.0,
"epoch": 0.5176,
"grad_norm": 7.342215538024902,
"kl": 2.76806640625,
"learning_rate": 1e-06,
"loss": 0.1712,
"num_tokens": 8625361.0,
"reward": -2.8857421875,
"reward_std": 11.833799362182617,
"rewards/rm_reward_func/mean": -2.8857421875,
"rewards/rm_reward_func/std": 15.419763565063477,
"step": 647
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 392.90625,
"completions/mean_terminated_length": 321.45001220703125,
"completions/min_length": 172.0,
"completions/min_terminated_length": 172.0,
"epoch": 0.5184,
"grad_norm": 16.402130126953125,
"kl": 4.22412109375,
"learning_rate": 1e-06,
"loss": 0.1936,
"num_tokens": 8639934.0,
"reward": 0.00299072265625,
"reward_std": 5.563260078430176,
"rewards/rm_reward_func/mean": 0.00299072265625,
"rewards/rm_reward_func/std": 17.13874053955078,
"step": 648
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 484.0,
"completions/mean_length": 174.40625,
"completions/mean_terminated_length": 163.51612854003906,
"completions/min_length": 24.0,
"completions/min_terminated_length": 24.0,
"epoch": 0.5192,
"grad_norm": 8.888167381286621,
"kl": 0.8353271484375,
"learning_rate": 1e-06,
"loss": -0.0946,
"num_tokens": 8652491.0,
"reward": 0.316680908203125,
"reward_std": 4.313920021057129,
"rewards/rm_reward_func/mean": 0.316680908203125,
"rewards/rm_reward_func/std": 10.923198699951172,
"step": 649
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 491.0,
"completions/mean_length": 290.03125,
"completions/mean_terminated_length": 189.13636779785156,
"completions/min_length": 28.0,
"completions/min_terminated_length": 28.0,
"epoch": 0.52,
"grad_norm": 6.159624099731445,
"kl": 1.846435546875,
"learning_rate": 1e-06,
"loss": 0.007,
"num_tokens": 8665452.0,
"reward": 4.464111328125,
"reward_std": 7.780835151672363,
"rewards/rm_reward_func/mean": 4.464111328125,
"rewards/rm_reward_func/std": 10.548760414123535,
"step": 650
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 265.6875,
"completions/mean_terminated_length": 208.84616088867188,
"completions/min_length": 67.0,
"completions/min_terminated_length": 67.0,
"epoch": 0.5208,
"grad_norm": 11.240917205810547,
"kl": 1.3095703125,
"learning_rate": 1e-06,
"loss": 0.063,
"num_tokens": 8676746.0,
"reward": 2.96875,
"reward_std": 5.190629005432129,
"rewards/rm_reward_func/mean": 2.96875,
"rewards/rm_reward_func/std": 12.547528266906738,
"step": 651
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.4375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 453.0,
"completions/mean_length": 299.84375,
"completions/mean_terminated_length": 134.8333282470703,
"completions/min_length": 69.0,
"completions/min_terminated_length": 69.0,
"epoch": 0.5216,
"grad_norm": 9.644030570983887,
"kl": 0.29296875,
"learning_rate": 1e-06,
"loss": -0.0588,
"num_tokens": 8689357.0,
"reward": 0.384521484375,
"reward_std": 3.9550719261169434,
"rewards/rm_reward_func/mean": 0.384521484375,
"rewards/rm_reward_func/std": 12.727936744689941,
"step": 652
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 502.0,
"completions/mean_length": 347.9375,
"completions/mean_terminated_length": 317.5555725097656,
"completions/min_length": 65.0,
"completions/min_terminated_length": 65.0,
"epoch": 0.5224,
"grad_norm": 10.710493087768555,
"kl": 0.42578125,
"learning_rate": 1e-06,
"loss": 0.0054,
"num_tokens": 8702963.0,
"reward": 10.232421875,
"reward_std": 6.635782241821289,
"rewards/rm_reward_func/mean": 10.232421875,
"rewards/rm_reward_func/std": 9.094783782958984,
"step": 653
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 470.0,
"completions/mean_length": 320.875,
"completions/mean_terminated_length": 301.10345458984375,
"completions/min_length": 153.0,
"completions/min_terminated_length": 153.0,
"epoch": 0.5232,
"grad_norm": 6.3006792068481445,
"kl": 0.252197265625,
"learning_rate": 1e-06,
"loss": 0.0086,
"num_tokens": 8716151.0,
"reward": 8.225341796875,
"reward_std": 7.599928855895996,
"rewards/rm_reward_func/mean": 8.225341796875,
"rewards/rm_reward_func/std": 19.037904739379883,
"step": 654
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 375.875,
"completions/mean_terminated_length": 294.20001220703125,
"completions/min_length": 78.0,
"completions/min_terminated_length": 78.0,
"epoch": 0.524,
"grad_norm": 8.997413635253906,
"kl": 0.244140625,
"learning_rate": 1e-06,
"loss": 0.0777,
"num_tokens": 8732867.0,
"reward": 4.292236328125,
"reward_std": 7.305292129516602,
"rewards/rm_reward_func/mean": 4.292236328125,
"rewards/rm_reward_func/std": 11.710954666137695,
"step": 655
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 497.0,
"completions/mean_length": 357.53125,
"completions/mean_terminated_length": 297.08697509765625,
"completions/min_length": 68.0,
"completions/min_terminated_length": 68.0,
"epoch": 0.5248,
"grad_norm": 6.956607818603516,
"kl": 0.75,
"learning_rate": 1e-06,
"loss": 0.0366,
"num_tokens": 8747420.0,
"reward": 11.81103515625,
"reward_std": 6.258151054382324,
"rewards/rm_reward_func/mean": 11.81103515625,
"rewards/rm_reward_func/std": 15.444485664367676,
"step": 656
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 507.0,
"completions/max_terminated_length": 507.0,
"completions/mean_length": 360.75,
"completions/mean_terminated_length": 360.75,
"completions/min_length": 148.0,
"completions/min_terminated_length": 148.0,
"epoch": 0.5256,
"grad_norm": 6.72127103805542,
"kl": 0.552490234375,
"learning_rate": 1e-06,
"loss": 0.0153,
"num_tokens": 8760988.0,
"reward": 10.35595703125,
"reward_std": 5.374138832092285,
"rewards/rm_reward_func/mean": 10.35595703125,
"rewards/rm_reward_func/std": 6.271159648895264,
"step": 657
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 286.84375,
"completions/mean_terminated_length": 223.79998779296875,
"completions/min_length": 66.0,
"completions/min_terminated_length": 66.0,
"epoch": 0.5264,
"grad_norm": 7.279812335968018,
"kl": 0.7529296875,
"learning_rate": 1e-06,
"loss": 0.0554,
"num_tokens": 8772311.0,
"reward": 3.248779296875,
"reward_std": 4.097195625305176,
"rewards/rm_reward_func/mean": 3.248779296875,
"rewards/rm_reward_func/std": 11.073434829711914,
"step": 658
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.46875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 507.0,
"completions/mean_length": 444.15625,
"completions/mean_terminated_length": 384.29412841796875,
"completions/min_length": 264.0,
"completions/min_terminated_length": 264.0,
"epoch": 0.5272,
"grad_norm": 7.7773356437683105,
"kl": 0.760986328125,
"learning_rate": 1e-06,
"loss": -0.0174,
"num_tokens": 8788708.0,
"reward": -2.1377792358398438,
"reward_std": 4.090878486633301,
"rewards/rm_reward_func/mean": -2.1377792358398438,
"rewards/rm_reward_func/std": 10.053132057189941,
"step": 659
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 417.59375,
"completions/mean_terminated_length": 391.1600036621094,
"completions/min_length": 282.0,
"completions/min_terminated_length": 282.0,
"epoch": 0.528,
"grad_norm": 6.207853317260742,
"kl": 0.521240234375,
"learning_rate": 1e-06,
"loss": 0.0471,
"num_tokens": 8804327.0,
"reward": 3.51171875,
"reward_std": 6.670271873474121,
"rewards/rm_reward_func/mean": 3.51171875,
"rewards/rm_reward_func/std": 13.217809677124023,
"step": 660
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 511.0,
"completions/mean_length": 222.25,
"completions/mean_terminated_length": 141.1199951171875,
"completions/min_length": 68.0,
"completions/min_terminated_length": 68.0,
"epoch": 0.5288,
"grad_norm": 3.570781946182251,
"kl": 2.3623046875,
"learning_rate": 1e-06,
"loss": 0.0811,
"num_tokens": 8815471.0,
"reward": 3.49066162109375,
"reward_std": 3.8420963287353516,
"rewards/rm_reward_func/mean": 3.49066162109375,
"rewards/rm_reward_func/std": 9.580389976501465,
"step": 661
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 459.0,
"completions/mean_length": 334.34375,
"completions/mean_terminated_length": 275.125,
"completions/min_length": 84.0,
"completions/min_terminated_length": 84.0,
"epoch": 0.5296,
"grad_norm": 16.329147338867188,
"kl": 0.8173828125,
"learning_rate": 1e-06,
"loss": 0.0685,
"num_tokens": 8828386.0,
"reward": 1.876220703125,
"reward_std": 7.676386833190918,
"rewards/rm_reward_func/mean": 1.876220703125,
"rewards/rm_reward_func/std": 12.502883911132812,
"step": 662
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 501.0,
"completions/mean_length": 398.625,
"completions/mean_terminated_length": 330.6000061035156,
"completions/min_length": 90.0,
"completions/min_terminated_length": 90.0,
"epoch": 0.5304,
"grad_norm": 6.071362018585205,
"kl": 1.19775390625,
"learning_rate": 1e-06,
"loss": 0.1563,
"num_tokens": 8844302.0,
"reward": 1.69091796875,
"reward_std": 8.782705307006836,
"rewards/rm_reward_func/mean": 1.69091796875,
"rewards/rm_reward_func/std": 12.565754890441895,
"step": 663
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 501.0,
"completions/mean_length": 232.6875,
"completions/mean_terminated_length": 223.6774139404297,
"completions/min_length": 26.0,
"completions/min_terminated_length": 26.0,
"epoch": 0.5312,
"grad_norm": 11.870681762695312,
"kl": 4.712890625,
"learning_rate": 1e-06,
"loss": 0.2744,
"num_tokens": 8855052.0,
"reward": -1.134765625,
"reward_std": 4.481464385986328,
"rewards/rm_reward_func/mean": -1.134765625,
"rewards/rm_reward_func/std": 17.394426345825195,
"step": 664
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 505.0,
"completions/mean_length": 358.21875,
"completions/mean_terminated_length": 353.258056640625,
"completions/min_length": 195.0,
"completions/min_terminated_length": 195.0,
"epoch": 0.532,
"grad_norm": 6.449156284332275,
"kl": 2.161865234375,
"learning_rate": 1e-06,
"loss": 0.0442,
"num_tokens": 8868555.0,
"reward": 2.504180908203125,
"reward_std": 8.275325775146484,
"rewards/rm_reward_func/mean": 2.504180908203125,
"rewards/rm_reward_func/std": 15.419231414794922,
"step": 665
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 412.40625,
"completions/mean_terminated_length": 373.4347839355469,
"completions/min_length": 72.0,
"completions/min_terminated_length": 72.0,
"epoch": 0.5328,
"grad_norm": 15.121233940124512,
"kl": 4.3203125,
"learning_rate": 1e-06,
"loss": 0.0551,
"num_tokens": 8886344.0,
"reward": 3.4033203125,
"reward_std": 12.214666366577148,
"rewards/rm_reward_func/mean": 3.4033203125,
"rewards/rm_reward_func/std": 17.266735076904297,
"step": 666
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 504.0,
"completions/mean_length": 277.125,
"completions/mean_terminated_length": 222.92308044433594,
"completions/min_length": 92.0,
"completions/min_terminated_length": 92.0,
"epoch": 0.5336,
"grad_norm": 20.014434814453125,
"kl": 5.343505859375,
"learning_rate": 1e-06,
"loss": 0.1802,
"num_tokens": 8900948.0,
"reward": -5.2738800048828125,
"reward_std": 6.429419994354248,
"rewards/rm_reward_func/mean": -5.2738800048828125,
"rewards/rm_reward_func/std": 11.191460609436035,
"step": 667
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 503.0,
"completions/mean_length": 286.21875,
"completions/mean_terminated_length": 223.0,
"completions/min_length": 64.0,
"completions/min_terminated_length": 64.0,
"epoch": 0.5344,
"grad_norm": 46.5359001159668,
"kl": 16.328125,
"learning_rate": 1e-06,
"loss": 0.6287,
"num_tokens": 8912667.0,
"reward": -19.37646484375,
"reward_std": 8.034518241882324,
"rewards/rm_reward_func/mean": -19.37646484375,
"rewards/rm_reward_func/std": 11.764345169067383,
"step": 668
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 503.0,
"completions/mean_length": 374.09375,
"completions/mean_terminated_length": 354.39288330078125,
"completions/min_length": 143.0,
"completions/min_terminated_length": 143.0,
"epoch": 0.5352,
"grad_norm": 8.606017112731934,
"kl": 4.453125,
"learning_rate": 1e-06,
"loss": 0.1118,
"num_tokens": 8926590.0,
"reward": 5.37371826171875,
"reward_std": 12.284038543701172,
"rewards/rm_reward_func/mean": 5.37371826171875,
"rewards/rm_reward_func/std": 20.101335525512695,
"step": 669
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 490.0,
"completions/mean_length": 341.5625,
"completions/mean_terminated_length": 323.9310302734375,
"completions/min_length": 68.0,
"completions/min_terminated_length": 68.0,
"epoch": 0.536,
"grad_norm": 20.169708251953125,
"kl": 2.861328125,
"learning_rate": 1e-06,
"loss": 0.3412,
"num_tokens": 8941640.0,
"reward": 8.84033203125,
"reward_std": 7.4539666175842285,
"rewards/rm_reward_func/mean": 8.84033203125,
"rewards/rm_reward_func/std": 11.035738945007324,
"step": 670
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 498.0,
"completions/mean_length": 315.84375,
"completions/mean_terminated_length": 250.45834350585938,
"completions/min_length": 51.0,
"completions/min_terminated_length": 51.0,
"epoch": 0.5368,
"grad_norm": 8.888484001159668,
"kl": 1.6533203125,
"learning_rate": 1e-06,
"loss": -0.0764,
"num_tokens": 8955819.0,
"reward": 1.8431396484375,
"reward_std": 9.139184951782227,
"rewards/rm_reward_func/mean": 1.8431396484375,
"rewards/rm_reward_func/std": 12.700114250183105,
"step": 671
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 499.0,
"completions/mean_length": 273.5625,
"completions/mean_terminated_length": 265.8709716796875,
"completions/min_length": 52.0,
"completions/min_terminated_length": 52.0,
"epoch": 0.5376,
"grad_norm": 10.354080200195312,
"kl": 1.771484375,
"learning_rate": 1e-06,
"loss": 0.1823,
"num_tokens": 8966917.0,
"reward": 3.476806640625,
"reward_std": 4.19705867767334,
"rewards/rm_reward_func/mean": 3.476806640625,
"rewards/rm_reward_func/std": 17.924257278442383,
"step": 672
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 331.0,
"completions/mean_length": 100.0,
"completions/mean_terminated_length": 86.70967102050781,
"completions/min_length": 46.0,
"completions/min_terminated_length": 46.0,
"epoch": 0.5384,
"grad_norm": 28.584192276000977,
"kl": 4.08642578125,
"learning_rate": 1e-06,
"loss": 0.4196,
"num_tokens": 8974605.0,
"reward": 1.3251953125,
"reward_std": 7.180523872375488,
"rewards/rm_reward_func/mean": 1.3251953125,
"rewards/rm_reward_func/std": 14.241349220275879,
"step": 673
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 406.0,
"completions/mean_length": 226.21875,
"completions/mean_terminated_length": 130.95834350585938,
"completions/min_length": 44.0,
"completions/min_terminated_length": 44.0,
"epoch": 0.5392,
"grad_norm": 35.00379943847656,
"kl": 4.3994140625,
"learning_rate": 1e-06,
"loss": 0.1642,
"num_tokens": 8986524.0,
"reward": -5.6318359375,
"reward_std": 3.752939224243164,
"rewards/rm_reward_func/mean": -5.6318359375,
"rewards/rm_reward_func/std": 12.199244499206543,
"step": 674
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 456.0,
"completions/mean_length": 313.5625,
"completions/mean_terminated_length": 223.3636474609375,
"completions/min_length": 51.0,
"completions/min_terminated_length": 51.0,
"epoch": 0.54,
"grad_norm": 41.50772476196289,
"kl": 8.8515625,
"learning_rate": 1e-06,
"loss": 0.2941,
"num_tokens": 8999078.0,
"reward": -8.677978515625,
"reward_std": 4.428235054016113,
"rewards/rm_reward_func/mean": -8.677978515625,
"rewards/rm_reward_func/std": 14.32275104522705,
"step": 675
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 366.0,
"completions/mean_length": 183.375,
"completions/mean_terminated_length": 122.51851654052734,
"completions/min_length": 57.0,
"completions/min_terminated_length": 57.0,
"epoch": 0.5408,
"grad_norm": 7.0692219734191895,
"kl": 4.337890625,
"learning_rate": 1e-06,
"loss": 0.1818,
"num_tokens": 9007402.0,
"reward": -1.382568359375,
"reward_std": 6.2474846839904785,
"rewards/rm_reward_func/mean": -1.382568359375,
"rewards/rm_reward_func/std": 9.534759521484375,
"step": 676
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 376.6875,
"completions/mean_terminated_length": 357.3571472167969,
"completions/min_length": 101.0,
"completions/min_terminated_length": 101.0,
"epoch": 0.5416,
"grad_norm": 52.67711639404297,
"kl": 6.74609375,
"learning_rate": 1e-06,
"loss": 0.187,
"num_tokens": 9021656.0,
"reward": 0.373046875,
"reward_std": 13.075346946716309,
"rewards/rm_reward_func/mean": 0.373046875,
"rewards/rm_reward_func/std": 22.373414993286133,
"step": 677
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.40625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 492.0,
"completions/mean_length": 340.0625,
"completions/mean_terminated_length": 222.42105102539062,
"completions/min_length": 35.0,
"completions/min_terminated_length": 35.0,
"epoch": 0.5424,
"grad_norm": 8.724742889404297,
"kl": 2.99169921875,
"learning_rate": 1e-06,
"loss": 0.0752,
"num_tokens": 9037954.0,
"reward": 3.04296875,
"reward_std": 12.51136589050293,
"rewards/rm_reward_func/mean": 3.04296875,
"rewards/rm_reward_func/std": 18.065021514892578,
"step": 678
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 501.0,
"completions/mean_length": 190.28125,
"completions/mean_terminated_length": 144.32144165039062,
"completions/min_length": 26.0,
"completions/min_terminated_length": 26.0,
"epoch": 0.5432,
"grad_norm": 17.94440269470215,
"kl": 4.4853515625,
"learning_rate": 1e-06,
"loss": 0.4292,
"num_tokens": 9046651.0,
"reward": 0.7139892578125,
"reward_std": 3.5457606315612793,
"rewards/rm_reward_func/mean": 0.7139892578125,
"rewards/rm_reward_func/std": 9.750553131103516,
"step": 679
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 474.0,
"completions/mean_length": 312.09375,
"completions/mean_terminated_length": 298.7666931152344,
"completions/min_length": 42.0,
"completions/min_terminated_length": 42.0,
"epoch": 0.544,
"grad_norm": 21.662830352783203,
"kl": 7.321533203125,
"learning_rate": 1e-06,
"loss": 0.1919,
"num_tokens": 9059910.0,
"reward": -3.553955078125,
"reward_std": 10.071688652038574,
"rewards/rm_reward_func/mean": -3.553955078125,
"rewards/rm_reward_func/std": 14.8821382522583,
"step": 680
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 505.0,
"completions/mean_length": 266.25,
"completions/mean_terminated_length": 220.74073791503906,
"completions/min_length": 72.0,
"completions/min_terminated_length": 72.0,
"epoch": 0.5448,
"grad_norm": 6.978208065032959,
"kl": 4.47412109375,
"learning_rate": 1e-06,
"loss": 0.1302,
"num_tokens": 9070510.0,
"reward": -1.48779296875,
"reward_std": 9.718238830566406,
"rewards/rm_reward_func/mean": -1.48779296875,
"rewards/rm_reward_func/std": 15.631418228149414,
"step": 681
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 503.0,
"completions/mean_length": 252.6875,
"completions/mean_terminated_length": 215.6428680419922,
"completions/min_length": 75.0,
"completions/min_terminated_length": 75.0,
"epoch": 0.5456,
"grad_norm": 132.93753051757812,
"kl": 9.15625,
"learning_rate": 1e-06,
"loss": 0.2597,
"num_tokens": 9080252.0,
"reward": -10.84326171875,
"reward_std": 9.151596069335938,
"rewards/rm_reward_func/mean": -10.84326171875,
"rewards/rm_reward_func/std": 14.070874214172363,
"step": 682
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 434.0,
"completions/mean_length": 145.9375,
"completions/mean_terminated_length": 121.53334045410156,
"completions/min_length": 27.0,
"completions/min_terminated_length": 27.0,
"epoch": 0.5464,
"grad_norm": 22.839073181152344,
"kl": 5.984375,
"learning_rate": 1e-06,
"loss": 0.1499,
"num_tokens": 9088962.0,
"reward": -9.11358642578125,
"reward_std": 12.002496719360352,
"rewards/rm_reward_func/mean": -9.11358642578125,
"rewards/rm_reward_func/std": 16.55123519897461,
"step": 683
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 475.0,
"completions/mean_length": 238.875,
"completions/mean_terminated_length": 210.6206817626953,
"completions/min_length": 26.0,
"completions/min_terminated_length": 26.0,
"epoch": 0.5472,
"grad_norm": 37.41764831542969,
"kl": 5.50390625,
"learning_rate": 1e-06,
"loss": 0.0492,
"num_tokens": 9098534.0,
"reward": -3.8291015625,
"reward_std": 8.362621307373047,
"rewards/rm_reward_func/mean": -3.8291015625,
"rewards/rm_reward_func/std": 15.430756568908691,
"step": 684
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 464.0,
"completions/mean_length": 230.5625,
"completions/mean_terminated_length": 221.48387145996094,
"completions/min_length": 23.0,
"completions/min_terminated_length": 23.0,
"epoch": 0.548,
"grad_norm": 19.959774017333984,
"kl": 5.6953125,
"learning_rate": 1e-06,
"loss": 0.0816,
"num_tokens": 9108744.0,
"reward": -6.86328125,
"reward_std": 15.26154613494873,
"rewards/rm_reward_func/mean": -6.86328125,
"rewards/rm_reward_func/std": 18.045209884643555,
"step": 685
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 505.0,
"completions/mean_length": 371.625,
"completions/mean_terminated_length": 345.629638671875,
"completions/min_length": 109.0,
"completions/min_terminated_length": 109.0,
"epoch": 0.5488,
"grad_norm": 10.441740989685059,
"kl": 4.70703125,
"learning_rate": 1e-06,
"loss": 0.2121,
"num_tokens": 9123308.0,
"reward": 0.5693359375,
"reward_std": 7.536800384521484,
"rewards/rm_reward_func/mean": 0.5693359375,
"rewards/rm_reward_func/std": 17.190340042114258,
"step": 686
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 507.0,
"completions/mean_length": 184.9375,
"completions/mean_terminated_length": 151.10345458984375,
"completions/min_length": 65.0,
"completions/min_terminated_length": 65.0,
"epoch": 0.5496,
"grad_norm": 12.261507987976074,
"kl": 2.86083984375,
"learning_rate": 1e-06,
"loss": 0.0786,
"num_tokens": 9132018.0,
"reward": 5.0924072265625,
"reward_std": 3.7133607864379883,
"rewards/rm_reward_func/mean": 5.0924072265625,
"rewards/rm_reward_func/std": 8.903958320617676,
"step": 687
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 493.0,
"completions/max_terminated_length": 493.0,
"completions/mean_length": 254.96875,
"completions/mean_terminated_length": 254.96875,
"completions/min_length": 8.0,
"completions/min_terminated_length": 8.0,
"epoch": 0.5504,
"grad_norm": 11.390403747558594,
"kl": 3.8935546875,
"learning_rate": 1e-06,
"loss": -0.1216,
"num_tokens": 9143177.0,
"reward": -8.523193359375,
"reward_std": 11.859602928161621,
"rewards/rm_reward_func/mean": -8.523193359375,
"rewards/rm_reward_func/std": 12.442675590515137,
"step": 688
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 407.0,
"completions/mean_length": 254.90625,
"completions/mean_terminated_length": 237.7666778564453,
"completions/min_length": 19.0,
"completions/min_terminated_length": 19.0,
"epoch": 0.5512,
"grad_norm": 9.756564140319824,
"kl": 4.615234375,
"learning_rate": 1e-06,
"loss": 0.1961,
"num_tokens": 9155374.0,
"reward": -8.00146484375,
"reward_std": 10.193147659301758,
"rewards/rm_reward_func/mean": -8.00146484375,
"rewards/rm_reward_func/std": 11.081028938293457,
"step": 689
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 500.0,
"completions/mean_length": 344.90625,
"completions/mean_terminated_length": 279.5217590332031,
"completions/min_length": 75.0,
"completions/min_terminated_length": 75.0,
"epoch": 0.552,
"grad_norm": 12.181934356689453,
"kl": 3.39208984375,
"learning_rate": 1e-06,
"loss": 0.0899,
"num_tokens": 9170707.0,
"reward": -6.829345703125,
"reward_std": 5.473711013793945,
"rewards/rm_reward_func/mean": -6.829345703125,
"rewards/rm_reward_func/std": 10.798709869384766,
"step": 690
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 484.0,
"completions/mean_length": 332.0625,
"completions/mean_terminated_length": 320.0666809082031,
"completions/min_length": 89.0,
"completions/min_terminated_length": 89.0,
"epoch": 0.5528,
"grad_norm": 10.006911277770996,
"kl": 1.677978515625,
"learning_rate": 1e-06,
"loss": -0.0336,
"num_tokens": 9184165.0,
"reward": 3.05224609375,
"reward_std": 7.134219169616699,
"rewards/rm_reward_func/mean": 3.05224609375,
"rewards/rm_reward_func/std": 13.809707641601562,
"step": 691
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 485.0,
"completions/mean_length": 250.5625,
"completions/mean_terminated_length": 202.1481475830078,
"completions/min_length": 67.0,
"completions/min_terminated_length": 67.0,
"epoch": 0.5536,
"grad_norm": 6.326221942901611,
"kl": 0.51953125,
"learning_rate": 1e-06,
"loss": -0.0631,
"num_tokens": 9195959.0,
"reward": 12.587890625,
"reward_std": 4.309545040130615,
"rewards/rm_reward_func/mean": 12.587890625,
"rewards/rm_reward_func/std": 6.763890266418457,
"step": 692
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 474.0,
"completions/mean_length": 217.6875,
"completions/mean_terminated_length": 208.19354248046875,
"completions/min_length": 12.0,
"completions/min_terminated_length": 12.0,
"epoch": 0.5544,
"grad_norm": 15.060230255126953,
"kl": 2.18994140625,
"learning_rate": 1e-06,
"loss": 0.1014,
"num_tokens": 9206501.0,
"reward": -7.5948486328125,
"reward_std": 6.410397529602051,
"rewards/rm_reward_func/mean": -7.5948486328125,
"rewards/rm_reward_func/std": 9.005189895629883,
"step": 693
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 444.0,
"completions/mean_length": 206.0,
"completions/mean_terminated_length": 196.1290283203125,
"completions/min_length": 37.0,
"completions/min_terminated_length": 37.0,
"epoch": 0.5552,
"grad_norm": 12.398327827453613,
"kl": 1.31103515625,
"learning_rate": 1e-06,
"loss": -0.1765,
"num_tokens": 9218037.0,
"reward": -7.2659912109375,
"reward_std": 6.091501712799072,
"rewards/rm_reward_func/mean": -7.2659912109375,
"rewards/rm_reward_func/std": 11.163793563842773,
"step": 694
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 487.0,
"completions/mean_length": 345.84375,
"completions/mean_terminated_length": 322.1071472167969,
"completions/min_length": 155.0,
"completions/min_terminated_length": 155.0,
"epoch": 0.556,
"grad_norm": 6.776838779449463,
"kl": 2.171875,
"learning_rate": 1e-06,
"loss": 0.0345,
"num_tokens": 9231384.0,
"reward": -5.34033203125,
"reward_std": 5.828690052032471,
"rewards/rm_reward_func/mean": -5.34033203125,
"rewards/rm_reward_func/std": 6.941495418548584,
"step": 695
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 490.0,
"completions/mean_length": 266.84375,
"completions/mean_terminated_length": 221.44444274902344,
"completions/min_length": 74.0,
"completions/min_terminated_length": 74.0,
"epoch": 0.5568,
"grad_norm": 6.661186695098877,
"kl": 0.462158203125,
"learning_rate": 1e-06,
"loss": 0.0817,
"num_tokens": 9244219.0,
"reward": 3.683349609375,
"reward_std": 6.5144147872924805,
"rewards/rm_reward_func/mean": 3.683349609375,
"rewards/rm_reward_func/std": 10.859166145324707,
"step": 696
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 504.0,
"completions/mean_length": 407.0,
"completions/mean_terminated_length": 382.7692565917969,
"completions/min_length": 139.0,
"completions/min_terminated_length": 139.0,
"epoch": 0.5576,
"grad_norm": 16.05687141418457,
"kl": 0.77197265625,
"learning_rate": 1e-06,
"loss": -0.0237,
"num_tokens": 9260139.0,
"reward": 3.64599609375,
"reward_std": 6.319300651550293,
"rewards/rm_reward_func/mean": 3.64599609375,
"rewards/rm_reward_func/std": 14.248506546020508,
"step": 697
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 474.0,
"completions/mean_length": 395.09375,
"completions/mean_terminated_length": 349.34783935546875,
"completions/min_length": 71.0,
"completions/min_terminated_length": 71.0,
"epoch": 0.5584,
"grad_norm": 7.179235458374023,
"kl": 1.25146484375,
"learning_rate": 1e-06,
"loss": -0.0007,
"num_tokens": 9275854.0,
"reward": 3.7479248046875,
"reward_std": 7.614123821258545,
"rewards/rm_reward_func/mean": 3.7479248046875,
"rewards/rm_reward_func/std": 17.118871688842773,
"step": 698
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 320.0,
"completions/mean_length": 225.90625,
"completions/mean_terminated_length": 172.92593383789062,
"completions/min_length": 74.0,
"completions/min_terminated_length": 74.0,
"epoch": 0.5592,
"grad_norm": 17.386568069458008,
"kl": 0.8544921875,
"learning_rate": 1e-06,
"loss": -0.137,
"num_tokens": 9286667.0,
"reward": -11.1297607421875,
"reward_std": 6.347601890563965,
"rewards/rm_reward_func/mean": -11.1297607421875,
"rewards/rm_reward_func/std": 6.936679363250732,
"step": 699
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 433.0,
"completions/max_terminated_length": 433.0,
"completions/mean_length": 210.59375,
"completions/mean_terminated_length": 210.59375,
"completions/min_length": 15.0,
"completions/min_terminated_length": 15.0,
"epoch": 0.56,
"grad_norm": 13.88813591003418,
"kl": 1.533203125,
"learning_rate": 1e-06,
"loss": 0.082,
"num_tokens": 9296486.0,
"reward": -1.9166259765625,
"reward_std": 6.144941806793213,
"rewards/rm_reward_func/mean": -1.9166259765625,
"rewards/rm_reward_func/std": 11.042118072509766,
"step": 700
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 417.0,
"completions/mean_length": 235.3125,
"completions/mean_terminated_length": 216.86668395996094,
"completions/min_length": 51.0,
"completions/min_terminated_length": 51.0,
"epoch": 0.5608,
"grad_norm": 14.585112571716309,
"kl": 1.34814453125,
"learning_rate": 1e-06,
"loss": -0.0184,
"num_tokens": 9306088.0,
"reward": -1.9686279296875,
"reward_std": 6.677350997924805,
"rewards/rm_reward_func/mean": -1.9686279296875,
"rewards/rm_reward_func/std": 8.160258293151855,
"step": 701
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 452.0,
"completions/mean_length": 243.34375,
"completions/mean_terminated_length": 225.433349609375,
"completions/min_length": 65.0,
"completions/min_terminated_length": 65.0,
"epoch": 0.5616,
"grad_norm": 16.384502410888672,
"kl": 1.71630859375,
"learning_rate": 1e-06,
"loss": 0.2464,
"num_tokens": 9319155.0,
"reward": 0.691162109375,
"reward_std": 6.911756992340088,
"rewards/rm_reward_func/mean": 0.691162109375,
"rewards/rm_reward_func/std": 17.84705924987793,
"step": 702
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 382.0,
"completions/max_terminated_length": 382.0,
"completions/mean_length": 148.78125,
"completions/mean_terminated_length": 148.78125,
"completions/min_length": 22.0,
"completions/min_terminated_length": 22.0,
"epoch": 0.5624,
"grad_norm": 6.983405113220215,
"kl": 0.526123046875,
"learning_rate": 1e-06,
"loss": -0.1855,
"num_tokens": 9326356.0,
"reward": -2.7777099609375,
"reward_std": 3.3739712238311768,
"rewards/rm_reward_func/mean": -2.7777099609375,
"rewards/rm_reward_func/std": 7.716010570526123,
"step": 703
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 476.0,
"completions/mean_length": 366.71875,
"completions/mean_terminated_length": 357.0333557128906,
"completions/min_length": 142.0,
"completions/min_terminated_length": 142.0,
"epoch": 0.5632,
"grad_norm": 10.195340156555176,
"kl": 1.0078125,
"learning_rate": 1e-06,
"loss": -0.0249,
"num_tokens": 9340835.0,
"reward": 5.092041015625,
"reward_std": 8.344895362854004,
"rewards/rm_reward_func/mean": 5.092041015625,
"rewards/rm_reward_func/std": 13.963737487792969,
"step": 704
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 430.0,
"completions/mean_length": 267.59375,
"completions/mean_terminated_length": 259.70965576171875,
"completions/min_length": 61.0,
"completions/min_terminated_length": 61.0,
"epoch": 0.564,
"grad_norm": 15.919843673706055,
"kl": 1.554931640625,
"learning_rate": 1e-06,
"loss": 0.159,
"num_tokens": 9352078.0,
"reward": -2.619140625,
"reward_std": 7.604259490966797,
"rewards/rm_reward_func/mean": -2.619140625,
"rewards/rm_reward_func/std": 10.991021156311035,
"step": 705
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 407.375,
"completions/mean_terminated_length": 366.4347839355469,
"completions/min_length": 95.0,
"completions/min_terminated_length": 95.0,
"epoch": 0.5648,
"grad_norm": 12.950475692749023,
"kl": 2.766845703125,
"learning_rate": 1e-06,
"loss": 0.068,
"num_tokens": 9367250.0,
"reward": 2.364013671875,
"reward_std": 8.381866455078125,
"rewards/rm_reward_func/mean": 2.364013671875,
"rewards/rm_reward_func/std": 12.802971839904785,
"step": 706
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 298.75,
"completions/mean_terminated_length": 259.2592468261719,
"completions/min_length": 72.0,
"completions/min_terminated_length": 72.0,
"epoch": 0.5656,
"grad_norm": 6.933451175689697,
"kl": 0.818603515625,
"learning_rate": 1e-06,
"loss": 0.0176,
"num_tokens": 9379698.0,
"reward": 4.7874755859375,
"reward_std": 7.199830532073975,
"rewards/rm_reward_func/mean": 4.7874755859375,
"rewards/rm_reward_func/std": 9.206077575683594,
"step": 707
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 441.0,
"completions/max_terminated_length": 441.0,
"completions/mean_length": 232.8125,
"completions/mean_terminated_length": 232.8125,
"completions/min_length": 72.0,
"completions/min_terminated_length": 72.0,
"epoch": 0.5664,
"grad_norm": 6.320394992828369,
"kl": 0.52197265625,
"learning_rate": 1e-06,
"loss": -0.1093,
"num_tokens": 9389684.0,
"reward": 13.0057373046875,
"reward_std": 8.226985931396484,
"rewards/rm_reward_func/mean": 13.0057373046875,
"rewards/rm_reward_func/std": 11.692206382751465,
"step": 708
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 468.0,
"completions/mean_length": 290.46875,
"completions/mean_terminated_length": 275.70001220703125,
"completions/min_length": 63.0,
"completions/min_terminated_length": 63.0,
"epoch": 0.5672,
"grad_norm": 9.328205108642578,
"kl": 2.07763671875,
"learning_rate": 1e-06,
"loss": 0.1139,
"num_tokens": 9400939.0,
"reward": 2.2265625,
"reward_std": 8.862954139709473,
"rewards/rm_reward_func/mean": 2.2265625,
"rewards/rm_reward_func/std": 12.553840637207031,
"step": 709
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 490.0,
"completions/mean_length": 286.625,
"completions/mean_terminated_length": 223.51998901367188,
"completions/min_length": 72.0,
"completions/min_terminated_length": 72.0,
"epoch": 0.568,
"grad_norm": 12.409536361694336,
"kl": 4.6474609375,
"learning_rate": 1e-06,
"loss": 0.2891,
"num_tokens": 9412575.0,
"reward": 3.2001953125,
"reward_std": 6.288336753845215,
"rewards/rm_reward_func/mean": 3.2001953125,
"rewards/rm_reward_func/std": 10.550227165222168,
"step": 710
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 511.0,
"completions/mean_length": 338.90625,
"completions/mean_terminated_length": 314.1785888671875,
"completions/min_length": 72.0,
"completions/min_terminated_length": 72.0,
"epoch": 0.5688,
"grad_norm": 5.891862869262695,
"kl": 1.07666015625,
"learning_rate": 1e-06,
"loss": 0.0507,
"num_tokens": 9427844.0,
"reward": 7.5908203125,
"reward_std": 3.023407459259033,
"rewards/rm_reward_func/mean": 7.5908203125,
"rewards/rm_reward_func/std": 12.348587036132812,
"step": 711
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 501.0,
"completions/mean_length": 338.59375,
"completions/mean_terminated_length": 259.7727355957031,
"completions/min_length": 48.0,
"completions/min_terminated_length": 48.0,
"epoch": 0.5696,
"grad_norm": 5.630682468414307,
"kl": 2.202880859375,
"learning_rate": 1e-06,
"loss": 0.0399,
"num_tokens": 9441263.0,
"reward": 4.940185546875,
"reward_std": 8.606001853942871,
"rewards/rm_reward_func/mean": 4.940185546875,
"rewards/rm_reward_func/std": 15.963677406311035,
"step": 712
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.59375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 475.0,
"completions/mean_length": 381.65625,
"completions/mean_terminated_length": 191.1538543701172,
"completions/min_length": 74.0,
"completions/min_terminated_length": 74.0,
"epoch": 0.5704,
"grad_norm": 21.119258880615234,
"kl": 10.06396484375,
"learning_rate": 1e-06,
"loss": 0.3982,
"num_tokens": 9456948.0,
"reward": -4.12841796875,
"reward_std": 10.323966026306152,
"rewards/rm_reward_func/mean": -4.12841796875,
"rewards/rm_reward_func/std": 16.72847557067871,
"step": 713
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.4375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 507.0,
"completions/mean_length": 382.0625,
"completions/mean_terminated_length": 281.0,
"completions/min_length": 67.0,
"completions/min_terminated_length": 67.0,
"epoch": 0.5712,
"grad_norm": 7.876568794250488,
"kl": 5.89892578125,
"learning_rate": 1e-06,
"loss": 0.1956,
"num_tokens": 9471470.0,
"reward": -1.296875,
"reward_std": 11.687822341918945,
"rewards/rm_reward_func/mean": -1.296875,
"rewards/rm_reward_func/std": 17.725196838378906,
"step": 714
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 301.6875,
"completions/mean_terminated_length": 219.3913116455078,
"completions/min_length": 72.0,
"completions/min_terminated_length": 72.0,
"epoch": 0.572,
"grad_norm": 16.49651527404785,
"kl": 6.4150390625,
"learning_rate": 1e-06,
"loss": 0.2308,
"num_tokens": 9486132.0,
"reward": -4.572021484375,
"reward_std": 6.965249061584473,
"rewards/rm_reward_func/mean": -4.572021484375,
"rewards/rm_reward_func/std": 12.301454544067383,
"step": 715
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 493.0,
"completions/mean_length": 383.8125,
"completions/mean_terminated_length": 306.8999938964844,
"completions/min_length": 72.0,
"completions/min_terminated_length": 72.0,
"epoch": 0.5728,
"grad_norm": 7.270954608917236,
"kl": 4.827392578125,
"learning_rate": 1e-06,
"loss": 0.2095,
"num_tokens": 9503854.0,
"reward": 20.08734130859375,
"reward_std": 11.994423866271973,
"rewards/rm_reward_func/mean": 20.08734130859375,
"rewards/rm_reward_func/std": 29.330028533935547,
"step": 716
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 394.40625,
"completions/mean_terminated_length": 361.47998046875,
"completions/min_length": 67.0,
"completions/min_terminated_length": 67.0,
"epoch": 0.5736,
"grad_norm": 5.99873161315918,
"kl": 5.353271484375,
"learning_rate": 1e-06,
"loss": 0.1456,
"num_tokens": 9519059.0,
"reward": 0.1048583984375,
"reward_std": 15.651195526123047,
"rewards/rm_reward_func/mean": 0.1048583984375,
"rewards/rm_reward_func/std": 15.79875373840332,
"step": 717
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 503.0,
"completions/mean_length": 340.53125,
"completions/mean_terminated_length": 300.9615478515625,
"completions/min_length": 71.0,
"completions/min_terminated_length": 71.0,
"epoch": 0.5744,
"grad_norm": 10.691807746887207,
"kl": 6.0263671875,
"learning_rate": 1e-06,
"loss": 0.1929,
"num_tokens": 9532652.0,
"reward": 3.922882080078125,
"reward_std": 8.720820426940918,
"rewards/rm_reward_func/mean": 3.922882080078125,
"rewards/rm_reward_func/std": 20.64432716369629,
"step": 718
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 475.0,
"completions/mean_length": 238.09375,
"completions/mean_terminated_length": 161.39999389648438,
"completions/min_length": 72.0,
"completions/min_terminated_length": 72.0,
"epoch": 0.5752,
"grad_norm": 5.596479892730713,
"kl": 2.0068359375,
"learning_rate": 1e-06,
"loss": 0.0605,
"num_tokens": 9545783.0,
"reward": 6.065826416015625,
"reward_std": 3.9478113651275635,
"rewards/rm_reward_func/mean": 6.065826416015625,
"rewards/rm_reward_func/std": 7.903535842895508,
"step": 719
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 393.5,
"completions/mean_terminated_length": 381.2413635253906,
"completions/min_length": 209.0,
"completions/min_terminated_length": 209.0,
"epoch": 0.576,
"grad_norm": 7.530872344970703,
"kl": 4.125,
"learning_rate": 1e-06,
"loss": 0.0967,
"num_tokens": 9561407.0,
"reward": 8.341751098632812,
"reward_std": 13.022964477539062,
"rewards/rm_reward_func/mean": 8.341751098632812,
"rewards/rm_reward_func/std": 16.08083724975586,
"step": 720
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 481.0,
"completions/mean_length": 346.71875,
"completions/mean_terminated_length": 260.1428527832031,
"completions/min_length": 65.0,
"completions/min_terminated_length": 65.0,
"epoch": 0.5768,
"grad_norm": 12.534733772277832,
"kl": 4.27685546875,
"learning_rate": 1e-06,
"loss": 0.1581,
"num_tokens": 9574310.0,
"reward": -3.28759765625,
"reward_std": 5.179610252380371,
"rewards/rm_reward_func/mean": -3.28759765625,
"rewards/rm_reward_func/std": 12.57066822052002,
"step": 721
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 365.28125,
"completions/mean_terminated_length": 307.86956787109375,
"completions/min_length": 74.0,
"completions/min_terminated_length": 74.0,
"epoch": 0.5776,
"grad_norm": 8.022375106811523,
"kl": 1.400390625,
"learning_rate": 1e-06,
"loss": 0.0708,
"num_tokens": 9589439.0,
"reward": 5.868088722229004,
"reward_std": 7.8726677894592285,
"rewards/rm_reward_func/mean": 5.868088722229004,
"rewards/rm_reward_func/std": 16.735122680664062,
"step": 722
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 500.0,
"completions/mean_length": 419.90625,
"completions/mean_terminated_length": 406.7500305175781,
"completions/min_length": 284.0,
"completions/min_terminated_length": 284.0,
"epoch": 0.5784,
"grad_norm": 6.199563980102539,
"kl": 1.158935546875,
"learning_rate": 1e-06,
"loss": 0.0303,
"num_tokens": 9604860.0,
"reward": 7.0489501953125,
"reward_std": 6.722829818725586,
"rewards/rm_reward_func/mean": 7.0489501953125,
"rewards/rm_reward_func/std": 10.408683776855469,
"step": 723
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 505.0,
"completions/mean_length": 410.5625,
"completions/mean_terminated_length": 364.4545593261719,
"completions/min_length": 104.0,
"completions/min_terminated_length": 104.0,
"epoch": 0.5792,
"grad_norm": 8.32470703125,
"kl": 2.0205078125,
"learning_rate": 1e-06,
"loss": 0.0497,
"num_tokens": 9620638.0,
"reward": 8.731201171875,
"reward_std": 13.764381408691406,
"rewards/rm_reward_func/mean": 8.731201171875,
"rewards/rm_reward_func/std": 16.106142044067383,
"step": 724
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 511.0,
"completions/mean_length": 333.1875,
"completions/mean_terminated_length": 239.52381896972656,
"completions/min_length": 59.0,
"completions/min_terminated_length": 59.0,
"epoch": 0.58,
"grad_norm": 9.582738876342773,
"kl": 2.6728515625,
"learning_rate": 1e-06,
"loss": 0.1381,
"num_tokens": 9633836.0,
"reward": 0.17033767700195312,
"reward_std": 8.874990463256836,
"rewards/rm_reward_func/mean": 0.17033767700195312,
"rewards/rm_reward_func/std": 9.867877960205078,
"step": 725
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 507.0,
"completions/mean_length": 358.40625,
"completions/mean_terminated_length": 277.952392578125,
"completions/min_length": 76.0,
"completions/min_terminated_length": 76.0,
"epoch": 0.5808,
"grad_norm": 5.19415807723999,
"kl": 2.4541015625,
"learning_rate": 1e-06,
"loss": 0.138,
"num_tokens": 9647841.0,
"reward": 3.939697265625,
"reward_std": 5.236965179443359,
"rewards/rm_reward_func/mean": 3.939697265625,
"rewards/rm_reward_func/std": 12.523097038269043,
"step": 726
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.40625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 507.0,
"completions/mean_length": 385.40625,
"completions/mean_terminated_length": 298.78948974609375,
"completions/min_length": 68.0,
"completions/min_terminated_length": 68.0,
"epoch": 0.5816,
"grad_norm": 7.742301940917969,
"kl": 2.68505859375,
"learning_rate": 1e-06,
"loss": 0.2645,
"num_tokens": 9663086.0,
"reward": 1.2738037109375,
"reward_std": 6.155429363250732,
"rewards/rm_reward_func/mean": 1.2738037109375,
"rewards/rm_reward_func/std": 7.7377028465271,
"step": 727
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 158.0,
"completions/max_terminated_length": 158.0,
"completions/mean_length": 79.90625,
"completions/mean_terminated_length": 79.90625,
"completions/min_length": 39.0,
"completions/min_terminated_length": 39.0,
"epoch": 0.5824,
"grad_norm": 7.791232585906982,
"kl": 1.28955078125,
"learning_rate": 1e-06,
"loss": 0.0041,
"num_tokens": 9670771.0,
"reward": -0.384521484375,
"reward_std": 2.3221030235290527,
"rewards/rm_reward_func/mean": -0.384521484375,
"rewards/rm_reward_func/std": 5.91448974609375,
"step": 728
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.40625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 503.0,
"completions/mean_length": 437.34375,
"completions/mean_terminated_length": 386.2631530761719,
"completions/min_length": 70.0,
"completions/min_terminated_length": 70.0,
"epoch": 0.5832,
"grad_norm": 6.162810325622559,
"kl": 2.515625,
"learning_rate": 1e-06,
"loss": 0.0724,
"num_tokens": 9687254.0,
"reward": 4.739013671875,
"reward_std": 13.861799240112305,
"rewards/rm_reward_func/mean": 4.739013671875,
"rewards/rm_reward_func/std": 15.850342750549316,
"step": 729
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 445.0,
"completions/mean_length": 365.15625,
"completions/mean_terminated_length": 307.6956481933594,
"completions/min_length": 105.0,
"completions/min_terminated_length": 105.0,
"epoch": 0.584,
"grad_norm": 8.422097206115723,
"kl": 3.58056640625,
"learning_rate": 1e-06,
"loss": 0.0715,
"num_tokens": 9700803.0,
"reward": -7.97607421875,
"reward_std": 8.740215301513672,
"rewards/rm_reward_func/mean": -7.97607421875,
"rewards/rm_reward_func/std": 10.935211181640625,
"step": 730
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 473.875,
"completions/mean_terminated_length": 451.0,
"completions/min_length": 347.0,
"completions/min_terminated_length": 347.0,
"epoch": 0.5848,
"grad_norm": 9.889822006225586,
"kl": 0.578125,
"learning_rate": 1e-06,
"loss": 0.0484,
"num_tokens": 9718783.0,
"reward": 15.586181640625,
"reward_std": 11.35405445098877,
"rewards/rm_reward_func/mean": 15.586181640625,
"rewards/rm_reward_func/std": 11.300307273864746,
"step": 731
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 479.0,
"completions/mean_length": 292.59375,
"completions/mean_terminated_length": 269.89654541015625,
"completions/min_length": 74.0,
"completions/min_terminated_length": 74.0,
"epoch": 0.5856,
"grad_norm": 7.809492111206055,
"kl": 0.862060546875,
"learning_rate": 1e-06,
"loss": 0.0524,
"num_tokens": 9732722.0,
"reward": 8.6011962890625,
"reward_std": 3.8772120475769043,
"rewards/rm_reward_func/mean": 8.6011962890625,
"rewards/rm_reward_func/std": 7.693234920501709,
"step": 732
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.40625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 503.0,
"completions/mean_length": 441.21875,
"completions/mean_terminated_length": 392.78948974609375,
"completions/min_length": 259.0,
"completions/min_terminated_length": 259.0,
"epoch": 0.5864,
"grad_norm": 7.425821781158447,
"kl": 3.7294921875,
"learning_rate": 1e-06,
"loss": 0.1576,
"num_tokens": 9749089.0,
"reward": 9.70703125,
"reward_std": 12.684297561645508,
"rewards/rm_reward_func/mean": 9.70703125,
"rewards/rm_reward_func/std": 14.209367752075195,
"step": 733
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 470.0,
"completions/mean_length": 310.65625,
"completions/mean_terminated_length": 231.86956787109375,
"completions/min_length": 74.0,
"completions/min_terminated_length": 74.0,
"epoch": 0.5872,
"grad_norm": 24.46310806274414,
"kl": 5.77197265625,
"learning_rate": 1e-06,
"loss": 0.2606,
"num_tokens": 9761590.0,
"reward": -5.23046875,
"reward_std": 6.721203804016113,
"rewards/rm_reward_func/mean": -5.23046875,
"rewards/rm_reward_func/std": 17.831560134887695,
"step": 734
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 511.0,
"completions/mean_length": 351.03125,
"completions/mean_terminated_length": 288.0434875488281,
"completions/min_length": 74.0,
"completions/min_terminated_length": 74.0,
"epoch": 0.588,
"grad_norm": 4.2976603507995605,
"kl": 1.31982421875,
"learning_rate": 1e-06,
"loss": 0.0227,
"num_tokens": 9775831.0,
"reward": 10.992095947265625,
"reward_std": 7.566171646118164,
"rewards/rm_reward_func/mean": 10.992095947265625,
"rewards/rm_reward_func/std": 9.95695686340332,
"step": 735
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 489.0,
"completions/mean_length": 261.46875,
"completions/mean_terminated_length": 225.6785888671875,
"completions/min_length": 72.0,
"completions/min_terminated_length": 72.0,
"epoch": 0.5888,
"grad_norm": 7.958250999450684,
"kl": 2.50341796875,
"learning_rate": 1e-06,
"loss": 0.137,
"num_tokens": 9786494.0,
"reward": 0.07373046875,
"reward_std": 6.209069728851318,
"rewards/rm_reward_func/mean": 0.07373046875,
"rewards/rm_reward_func/std": 12.3759765625,
"step": 736
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 330.8125,
"completions/mean_terminated_length": 324.9677429199219,
"completions/min_length": 24.0,
"completions/min_terminated_length": 24.0,
"epoch": 0.5896,
"grad_norm": 5.071633815765381,
"kl": 1.260986328125,
"learning_rate": 1e-06,
"loss": -0.0572,
"num_tokens": 9799144.0,
"reward": 9.56390380859375,
"reward_std": 10.431290626525879,
"rewards/rm_reward_func/mean": 9.56390380859375,
"rewards/rm_reward_func/std": 19.839946746826172,
"step": 737
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 497.0,
"completions/mean_length": 380.25,
"completions/mean_terminated_length": 328.6956481933594,
"completions/min_length": 72.0,
"completions/min_terminated_length": 72.0,
"epoch": 0.5904,
"grad_norm": 10.128765106201172,
"kl": 3.7099609375,
"learning_rate": 1e-06,
"loss": 0.1282,
"num_tokens": 9816376.0,
"reward": 7.971923828125,
"reward_std": 5.655712127685547,
"rewards/rm_reward_func/mean": 7.971923828125,
"rewards/rm_reward_func/std": 19.648178100585938,
"step": 738
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 320.96875,
"completions/mean_terminated_length": 276.8846130371094,
"completions/min_length": 75.0,
"completions/min_terminated_length": 75.0,
"epoch": 0.5912,
"grad_norm": 12.91031551361084,
"kl": 1.35986328125,
"learning_rate": 1e-06,
"loss": 0.0801,
"num_tokens": 9832231.0,
"reward": 14.126220703125,
"reward_std": 8.189908981323242,
"rewards/rm_reward_func/mean": 14.126220703125,
"rewards/rm_reward_func/std": 11.697257995605469,
"step": 739
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 494.0,
"completions/mean_length": 344.125,
"completions/mean_terminated_length": 288.16668701171875,
"completions/min_length": 75.0,
"completions/min_terminated_length": 75.0,
"epoch": 0.592,
"grad_norm": 6.711599826812744,
"kl": 6.6611328125,
"learning_rate": 1e-06,
"loss": 0.4934,
"num_tokens": 9846835.0,
"reward": 3.853271484375,
"reward_std": 11.102495193481445,
"rewards/rm_reward_func/mean": 3.853271484375,
"rewards/rm_reward_func/std": 13.937686920166016,
"step": 740
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 507.0,
"completions/mean_length": 369.5,
"completions/mean_terminated_length": 294.8571472167969,
"completions/min_length": 42.0,
"completions/min_terminated_length": 42.0,
"epoch": 0.5928,
"grad_norm": 9.946474075317383,
"kl": 3.537353515625,
"learning_rate": 1e-06,
"loss": 0.2248,
"num_tokens": 9863235.0,
"reward": 5.8260498046875,
"reward_std": 8.534289360046387,
"rewards/rm_reward_func/mean": 5.8260498046875,
"rewards/rm_reward_func/std": 14.377345085144043,
"step": 741
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 487.0,
"completions/mean_length": 261.8125,
"completions/mean_terminated_length": 204.07693481445312,
"completions/min_length": 67.0,
"completions/min_terminated_length": 67.0,
"epoch": 0.5936,
"grad_norm": 34.741207122802734,
"kl": 16.79248046875,
"learning_rate": 1e-06,
"loss": 0.9002,
"num_tokens": 9875661.0,
"reward": -8.799308776855469,
"reward_std": 6.605631351470947,
"rewards/rm_reward_func/mean": -8.799308776855469,
"rewards/rm_reward_func/std": 15.544058799743652,
"step": 742
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 503.0,
"completions/mean_length": 390.09375,
"completions/mean_terminated_length": 316.95001220703125,
"completions/min_length": 100.0,
"completions/min_terminated_length": 100.0,
"epoch": 0.5944,
"grad_norm": 7.391372203826904,
"kl": 0.4296875,
"learning_rate": 1e-06,
"loss": 0.0217,
"num_tokens": 9891272.0,
"reward": 8.581817626953125,
"reward_std": 3.363267183303833,
"rewards/rm_reward_func/mean": 8.581817626953125,
"rewards/rm_reward_func/std": 12.903508186340332,
"step": 743
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 262.15625,
"completions/mean_terminated_length": 254.09677124023438,
"completions/min_length": 74.0,
"completions/min_terminated_length": 74.0,
"epoch": 0.5952,
"grad_norm": 6.578904151916504,
"kl": 5.13134765625,
"learning_rate": 1e-06,
"loss": 0.4582,
"num_tokens": 9904589.0,
"reward": 18.150390625,
"reward_std": 6.425145626068115,
"rewards/rm_reward_func/mean": 18.150390625,
"rewards/rm_reward_func/std": 10.988823890686035,
"step": 744
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 450.0,
"completions/mean_length": 300.03125,
"completions/mean_terminated_length": 251.11538696289062,
"completions/min_length": 34.0,
"completions/min_terminated_length": 34.0,
"epoch": 0.596,
"grad_norm": 26.791982650756836,
"kl": 5.4140625,
"learning_rate": 1e-06,
"loss": 0.1141,
"num_tokens": 9918158.0,
"reward": -5.0845947265625,
"reward_std": 7.70248556137085,
"rewards/rm_reward_func/mean": -5.0845947265625,
"rewards/rm_reward_func/std": 11.328791618347168,
"step": 745
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 461.0,
"completions/mean_length": 314.125,
"completions/mean_terminated_length": 268.4615478515625,
"completions/min_length": 75.0,
"completions/min_terminated_length": 75.0,
"epoch": 0.5968,
"grad_norm": 10.60112476348877,
"kl": 2.5537109375,
"learning_rate": 1e-06,
"loss": -0.0021,
"num_tokens": 9930418.0,
"reward": 3.98583984375,
"reward_std": 7.434886932373047,
"rewards/rm_reward_func/mean": 3.98583984375,
"rewards/rm_reward_func/std": 15.446136474609375,
"step": 746
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 511.0,
"completions/mean_length": 408.5625,
"completions/mean_terminated_length": 397.862060546875,
"completions/min_length": 251.0,
"completions/min_terminated_length": 251.0,
"epoch": 0.5976,
"grad_norm": 7.734127998352051,
"kl": 1.279541015625,
"learning_rate": 1e-06,
"loss": 0.0852,
"num_tokens": 9947396.0,
"reward": 8.797119140625,
"reward_std": 9.704774856567383,
"rewards/rm_reward_func/mean": 8.797119140625,
"rewards/rm_reward_func/std": 14.60503101348877,
"step": 747
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 499.0,
"completions/mean_length": 305.9375,
"completions/mean_terminated_length": 299.2903137207031,
"completions/min_length": 66.0,
"completions/min_terminated_length": 66.0,
"epoch": 0.5984,
"grad_norm": 30.35190200805664,
"kl": 10.0810546875,
"learning_rate": 1e-06,
"loss": 0.2479,
"num_tokens": 9959546.0,
"reward": -7.094024658203125,
"reward_std": 9.547418594360352,
"rewards/rm_reward_func/mean": -7.094024658203125,
"rewards/rm_reward_func/std": 12.000333786010742,
"step": 748
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 499.0,
"completions/mean_length": 285.0,
"completions/mean_terminated_length": 277.6773986816406,
"completions/min_length": 111.0,
"completions/min_terminated_length": 111.0,
"epoch": 0.5992,
"grad_norm": 9.810269355773926,
"kl": 0.603515625,
"learning_rate": 1e-06,
"loss": 0.057,
"num_tokens": 9970658.0,
"reward": 2.1044921875,
"reward_std": 3.5760817527770996,
"rewards/rm_reward_func/mean": 2.1044921875,
"rewards/rm_reward_func/std": 6.474411487579346,
"step": 749
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 503.0,
"completions/mean_length": 356.4375,
"completions/mean_terminated_length": 295.5652160644531,
"completions/min_length": 74.0,
"completions/min_terminated_length": 74.0,
"epoch": 0.6,
"grad_norm": 14.325955390930176,
"kl": 3.8134765625,
"learning_rate": 1e-06,
"loss": 0.0913,
"num_tokens": 9984464.0,
"reward": 4.713623046875,
"reward_std": 8.757728576660156,
"rewards/rm_reward_func/mean": 4.713623046875,
"rewards/rm_reward_func/std": 16.11822509765625,
"step": 750
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 297.5625,
"completions/mean_terminated_length": 275.3793029785156,
"completions/min_length": 74.0,
"completions/min_terminated_length": 74.0,
"epoch": 0.6008,
"grad_norm": 44.704063415527344,
"kl": 3.189453125,
"learning_rate": 1e-06,
"loss": 0.2095,
"num_tokens": 9997474.0,
"reward": 10.037109375,
"reward_std": 13.35189151763916,
"rewards/rm_reward_func/mean": 10.037109375,
"rewards/rm_reward_func/std": 21.836666107177734,
"step": 751
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 501.0,
"completions/mean_length": 264.59375,
"completions/mean_terminated_length": 256.6128845214844,
"completions/min_length": 77.0,
"completions/min_terminated_length": 77.0,
"epoch": 0.6016,
"grad_norm": 7.517630100250244,
"kl": 4.232177734375,
"learning_rate": 1e-06,
"loss": 0.215,
"num_tokens": 10008461.0,
"reward": 0.819580078125,
"reward_std": 10.651954650878906,
"rewards/rm_reward_func/mean": 0.819580078125,
"rewards/rm_reward_func/std": 15.250157356262207,
"step": 752
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 293.25,
"completions/mean_terminated_length": 278.66668701171875,
"completions/min_length": 77.0,
"completions/min_terminated_length": 77.0,
"epoch": 0.6024,
"grad_norm": 10.266236305236816,
"kl": 4.54296875,
"learning_rate": 1e-06,
"loss": 0.0359,
"num_tokens": 10020021.0,
"reward": -3.8173828125,
"reward_std": 9.468101501464844,
"rewards/rm_reward_func/mean": -3.8173828125,
"rewards/rm_reward_func/std": 13.068405151367188,
"step": 753
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 477.0,
"completions/mean_length": 192.4375,
"completions/mean_terminated_length": 146.7857208251953,
"completions/min_length": 42.0,
"completions/min_terminated_length": 42.0,
"epoch": 0.6032,
"grad_norm": 9.960123062133789,
"kl": 5.95947265625,
"learning_rate": 1e-06,
"loss": 0.0333,
"num_tokens": 10030379.0,
"reward": -3.7373046875,
"reward_std": 10.038860321044922,
"rewards/rm_reward_func/mean": -3.7373046875,
"rewards/rm_reward_func/std": 15.353011131286621,
"step": 754
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 463.0,
"completions/mean_length": 317.65625,
"completions/mean_terminated_length": 241.60870361328125,
"completions/min_length": 111.0,
"completions/min_terminated_length": 111.0,
"epoch": 0.604,
"grad_norm": 6.909921169281006,
"kl": 1.919921875,
"learning_rate": 1e-06,
"loss": 0.0276,
"num_tokens": 10043216.0,
"reward": 1.073760986328125,
"reward_std": 6.802268028259277,
"rewards/rm_reward_func/mean": 1.073760986328125,
"rewards/rm_reward_func/std": 10.724485397338867,
"step": 755
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 486.0,
"completions/mean_length": 322.78125,
"completions/mean_terminated_length": 310.16668701171875,
"completions/min_length": 77.0,
"completions/min_terminated_length": 77.0,
"epoch": 0.6048,
"grad_norm": 4.838641166687012,
"kl": 1.0595703125,
"learning_rate": 1e-06,
"loss": -0.0381,
"num_tokens": 10057593.0,
"reward": 16.49951171875,
"reward_std": 8.707597732543945,
"rewards/rm_reward_func/mean": 16.49951171875,
"rewards/rm_reward_func/std": 13.02664566040039,
"step": 756
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 232.6875,
"completions/mean_terminated_length": 223.6774139404297,
"completions/min_length": 29.0,
"completions/min_terminated_length": 29.0,
"epoch": 0.6056,
"grad_norm": 6.288021087646484,
"kl": 2.422119140625,
"learning_rate": 1e-06,
"loss": 0.0256,
"num_tokens": 10068959.0,
"reward": 7.3639373779296875,
"reward_std": 7.31492805480957,
"rewards/rm_reward_func/mean": 7.3639373779296875,
"rewards/rm_reward_func/std": 11.444160461425781,
"step": 757
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 461.0,
"completions/mean_length": 257.375,
"completions/mean_terminated_length": 240.40000915527344,
"completions/min_length": 24.0,
"completions/min_terminated_length": 24.0,
"epoch": 0.6064,
"grad_norm": 12.485620498657227,
"kl": 1.30322265625,
"learning_rate": 1e-06,
"loss": -0.256,
"num_tokens": 10079523.0,
"reward": 8.375,
"reward_std": 11.750480651855469,
"rewards/rm_reward_func/mean": 8.375,
"rewards/rm_reward_func/std": 15.094318389892578,
"step": 758
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 359.3125,
"completions/mean_terminated_length": 308.41668701171875,
"completions/min_length": 77.0,
"completions/min_terminated_length": 77.0,
"epoch": 0.6072,
"grad_norm": 67.01472473144531,
"kl": 0.41552734375,
"learning_rate": 1e-06,
"loss": -0.0451,
"num_tokens": 10096813.0,
"reward": 8.5546875,
"reward_std": 4.309829235076904,
"rewards/rm_reward_func/mean": 8.5546875,
"rewards/rm_reward_func/std": 15.106447219848633,
"step": 759
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 504.0,
"completions/mean_length": 387.09375,
"completions/mean_terminated_length": 352.1199951171875,
"completions/min_length": 169.0,
"completions/min_terminated_length": 169.0,
"epoch": 0.608,
"grad_norm": 7.828210830688477,
"kl": 1.158203125,
"learning_rate": 1e-06,
"loss": 0.0712,
"num_tokens": 10112304.0,
"reward": 2.0579833984375,
"reward_std": 6.627885341644287,
"rewards/rm_reward_func/mean": 2.0579833984375,
"rewards/rm_reward_func/std": 9.508418083190918,
"step": 760
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 299.96875,
"completions/mean_terminated_length": 240.59999084472656,
"completions/min_length": 21.0,
"completions/min_terminated_length": 21.0,
"epoch": 0.6088,
"grad_norm": 32.08409118652344,
"kl": 9.375,
"learning_rate": 1e-06,
"loss": 0.0282,
"num_tokens": 10124279.0,
"reward": -8.2109375,
"reward_std": 14.024812698364258,
"rewards/rm_reward_func/mean": -8.2109375,
"rewards/rm_reward_func/std": 16.273033142089844,
"step": 761
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 475.0,
"completions/mean_length": 267.03125,
"completions/mean_terminated_length": 185.375,
"completions/min_length": 50.0,
"completions/min_terminated_length": 50.0,
"epoch": 0.6096,
"grad_norm": 9.463338851928711,
"kl": 0.489013671875,
"learning_rate": 1e-06,
"loss": 0.0646,
"num_tokens": 10135368.0,
"reward": 4.3076171875,
"reward_std": 3.226278781890869,
"rewards/rm_reward_func/mean": 4.3076171875,
"rewards/rm_reward_func/std": 11.224862098693848,
"step": 762
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 210.40625,
"completions/mean_terminated_length": 210.40625,
"completions/min_length": 46.0,
"completions/min_terminated_length": 46.0,
"epoch": 0.6104,
"grad_norm": 15.222582817077637,
"kl": 2.01904296875,
"learning_rate": 1e-06,
"loss": 0.1217,
"num_tokens": 10144741.0,
"reward": -5.0877685546875,
"reward_std": 11.080349922180176,
"rewards/rm_reward_func/mean": -5.0877685546875,
"rewards/rm_reward_func/std": 13.31748104095459,
"step": 763
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 414.3125,
"completions/mean_terminated_length": 386.9599914550781,
"completions/min_length": 259.0,
"completions/min_terminated_length": 259.0,
"epoch": 0.6112,
"grad_norm": 13.568563461303711,
"kl": 3.62890625,
"learning_rate": 1e-06,
"loss": 0.0975,
"num_tokens": 10160167.0,
"reward": 6.3486328125,
"reward_std": 12.485017776489258,
"rewards/rm_reward_func/mean": 6.3486328125,
"rewards/rm_reward_func/std": 13.953908920288086,
"step": 764
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 378.0,
"completions/max_terminated_length": 378.0,
"completions/mean_length": 162.1875,
"completions/mean_terminated_length": 162.1875,
"completions/min_length": 77.0,
"completions/min_terminated_length": 77.0,
"epoch": 0.612,
"grad_norm": 19.0142765045166,
"kl": 2.60107421875,
"learning_rate": 1e-06,
"loss": 0.0385,
"num_tokens": 10170917.0,
"reward": -2.573974609375,
"reward_std": 3.245800495147705,
"rewards/rm_reward_func/mean": -2.573974609375,
"rewards/rm_reward_func/std": 13.304573059082031,
"step": 765
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.53125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 491.0,
"completions/mean_length": 436.625,
"completions/mean_terminated_length": 351.20001220703125,
"completions/min_length": 98.0,
"completions/min_terminated_length": 98.0,
"epoch": 0.6128,
"grad_norm": 24.08083724975586,
"kl": 2.4443359375,
"learning_rate": 1e-06,
"loss": 0.0583,
"num_tokens": 10190281.0,
"reward": 2.33367919921875,
"reward_std": 6.308701992034912,
"rewards/rm_reward_func/mean": 2.33367919921875,
"rewards/rm_reward_func/std": 15.869560241699219,
"step": 766
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 505.0,
"completions/mean_length": 345.46875,
"completions/mean_terminated_length": 280.3043518066406,
"completions/min_length": 65.0,
"completions/min_terminated_length": 65.0,
"epoch": 0.6136,
"grad_norm": 9.395527839660645,
"kl": 2.9775390625,
"learning_rate": 1e-06,
"loss": 0.0412,
"num_tokens": 10205904.0,
"reward": -1.7017822265625,
"reward_std": 6.9784321784973145,
"rewards/rm_reward_func/mean": -1.7017822265625,
"rewards/rm_reward_func/std": 11.290467262268066,
"step": 767
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 479.0,
"completions/mean_length": 262.09375,
"completions/mean_terminated_length": 131.1904754638672,
"completions/min_length": 77.0,
"completions/min_terminated_length": 77.0,
"epoch": 0.6144,
"grad_norm": 6.257786750793457,
"kl": 1.263671875,
"learning_rate": 1e-06,
"loss": -0.0265,
"num_tokens": 10219307.0,
"reward": 0.228759765625,
"reward_std": 3.712106466293335,
"rewards/rm_reward_func/mean": 0.228759765625,
"rewards/rm_reward_func/std": 7.532781600952148,
"step": 768
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 469.0,
"completions/mean_length": 219.875,
"completions/mean_terminated_length": 200.40000915527344,
"completions/min_length": 78.0,
"completions/min_terminated_length": 78.0,
"epoch": 0.6152,
"grad_norm": 12.303415298461914,
"kl": 1.0126953125,
"learning_rate": 1e-06,
"loss": -0.1391,
"num_tokens": 10229703.0,
"reward": 1.8739013671875,
"reward_std": 9.617498397827148,
"rewards/rm_reward_func/mean": 1.8739013671875,
"rewards/rm_reward_func/std": 12.91628360748291,
"step": 769
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 399.0625,
"completions/mean_terminated_length": 367.44000244140625,
"completions/min_length": 95.0,
"completions/min_terminated_length": 95.0,
"epoch": 0.616,
"grad_norm": 7.800823211669922,
"kl": 2.70703125,
"learning_rate": 1e-06,
"loss": 0.0288,
"num_tokens": 10244537.0,
"reward": 12.45703125,
"reward_std": 13.960775375366211,
"rewards/rm_reward_func/mean": 12.45703125,
"rewards/rm_reward_func/std": 23.239635467529297,
"step": 770
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 492.0,
"completions/mean_length": 414.78125,
"completions/mean_terminated_length": 376.7391357421875,
"completions/min_length": 184.0,
"completions/min_terminated_length": 184.0,
"epoch": 0.6168,
"grad_norm": 23.5693302154541,
"kl": 1.23095703125,
"learning_rate": 1e-06,
"loss": 0.0611,
"num_tokens": 10260738.0,
"reward": 7.40087890625,
"reward_std": 7.460350036621094,
"rewards/rm_reward_func/mean": 7.40087890625,
"rewards/rm_reward_func/std": 15.638055801391602,
"step": 771
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 442.0,
"completions/mean_length": 293.3125,
"completions/mean_terminated_length": 262.0714416503906,
"completions/min_length": 82.0,
"completions/min_terminated_length": 82.0,
"epoch": 0.6176,
"grad_norm": 24.51900291442871,
"kl": 1.25,
"learning_rate": 1e-06,
"loss": -0.0135,
"num_tokens": 10275532.0,
"reward": 7.166015625,
"reward_std": 10.221585273742676,
"rewards/rm_reward_func/mean": 7.166015625,
"rewards/rm_reward_func/std": 18.162363052368164,
"step": 772
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 448.0,
"completions/mean_length": 351.6875,
"completions/mean_terminated_length": 341.0000305175781,
"completions/min_length": 146.0,
"completions/min_terminated_length": 146.0,
"epoch": 0.6184,
"grad_norm": 11.453636169433594,
"kl": 0.533447265625,
"learning_rate": 1e-06,
"loss": -0.0055,
"num_tokens": 10288858.0,
"reward": 10.736896514892578,
"reward_std": 6.508322715759277,
"rewards/rm_reward_func/mean": 10.736896514892578,
"rewards/rm_reward_func/std": 15.74055290222168,
"step": 773
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 489.0,
"completions/mean_length": 187.375,
"completions/mean_terminated_length": 165.73333740234375,
"completions/min_length": 18.0,
"completions/min_terminated_length": 18.0,
"epoch": 0.6192,
"grad_norm": 11.30697250366211,
"kl": 2.5537109375,
"learning_rate": 1e-06,
"loss": 0.0897,
"num_tokens": 10299606.0,
"reward": -3.98828125,
"reward_std": 3.1018943786621094,
"rewards/rm_reward_func/mean": -3.98828125,
"rewards/rm_reward_func/std": 13.498686790466309,
"step": 774
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 507.0,
"completions/max_terminated_length": 507.0,
"completions/mean_length": 319.6875,
"completions/mean_terminated_length": 319.6875,
"completions/min_length": 84.0,
"completions/min_terminated_length": 84.0,
"epoch": 0.62,
"grad_norm": 13.508712768554688,
"kl": 5.896484375,
"learning_rate": 1e-06,
"loss": 0.2158,
"num_tokens": 10312068.0,
"reward": -2.490478515625,
"reward_std": 13.193625450134277,
"rewards/rm_reward_func/mean": -2.490478515625,
"rewards/rm_reward_func/std": 20.5009765625,
"step": 775
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 284.3125,
"completions/mean_terminated_length": 220.55999755859375,
"completions/min_length": 76.0,
"completions/min_terminated_length": 76.0,
"epoch": 0.6208,
"grad_norm": 30.201000213623047,
"kl": 3.12548828125,
"learning_rate": 1e-06,
"loss": 0.0509,
"num_tokens": 10324806.0,
"reward": 6.44482421875,
"reward_std": 9.712287902832031,
"rewards/rm_reward_func/mean": 6.44482421875,
"rewards/rm_reward_func/std": 14.30795669555664,
"step": 776
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 490.0,
"completions/mean_length": 299.1875,
"completions/mean_terminated_length": 292.32257080078125,
"completions/min_length": 40.0,
"completions/min_terminated_length": 40.0,
"epoch": 0.6216,
"grad_norm": 33.898277282714844,
"kl": 3.952392578125,
"learning_rate": 1e-06,
"loss": 0.0502,
"num_tokens": 10337852.0,
"reward": 0.22265625,
"reward_std": 9.971935272216797,
"rewards/rm_reward_func/mean": 0.22265625,
"rewards/rm_reward_func/std": 19.976945877075195,
"step": 777
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 470.0,
"completions/mean_length": 257.09375,
"completions/mean_terminated_length": 248.87095642089844,
"completions/min_length": 77.0,
"completions/min_terminated_length": 77.0,
"epoch": 0.6224,
"grad_norm": 7.3413543701171875,
"kl": 1.251708984375,
"learning_rate": 1e-06,
"loss": 0.005,
"num_tokens": 10350439.0,
"reward": 5.71923828125,
"reward_std": 5.034487724304199,
"rewards/rm_reward_func/mean": 5.71923828125,
"rewards/rm_reward_func/std": 12.428587913513184,
"step": 778
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 206.75,
"completions/mean_terminated_length": 196.90321350097656,
"completions/min_length": 20.0,
"completions/min_terminated_length": 20.0,
"epoch": 0.6232,
"grad_norm": 15.728623390197754,
"kl": 5.38427734375,
"learning_rate": 1e-06,
"loss": 0.185,
"num_tokens": 10360047.0,
"reward": -2.06689453125,
"reward_std": 5.258633613586426,
"rewards/rm_reward_func/mean": -2.06689453125,
"rewards/rm_reward_func/std": 14.16489315032959,
"step": 779
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 181.03125,
"completions/mean_terminated_length": 170.35482788085938,
"completions/min_length": 62.0,
"completions/min_terminated_length": 62.0,
"epoch": 0.624,
"grad_norm": 9.44076919555664,
"kl": 0.400390625,
"learning_rate": 1e-06,
"loss": 0.0051,
"num_tokens": 10369952.0,
"reward": 6.325927734375,
"reward_std": 2.282184600830078,
"rewards/rm_reward_func/mean": 6.325927734375,
"rewards/rm_reward_func/std": 5.712595462799072,
"step": 780
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 485.0,
"completions/mean_length": 298.90625,
"completions/mean_terminated_length": 292.0322570800781,
"completions/min_length": 42.0,
"completions/min_terminated_length": 42.0,
"epoch": 0.6248,
"grad_norm": 7.031983375549316,
"kl": 3.04150390625,
"learning_rate": 1e-06,
"loss": -0.0902,
"num_tokens": 10384941.0,
"reward": 12.3427734375,
"reward_std": 17.103315353393555,
"rewards/rm_reward_func/mean": 12.3427734375,
"rewards/rm_reward_func/std": 21.697189331054688,
"step": 781
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 504.0,
"completions/mean_length": 345.53125,
"completions/mean_terminated_length": 334.433349609375,
"completions/min_length": 104.0,
"completions/min_terminated_length": 104.0,
"epoch": 0.6256,
"grad_norm": 12.419363021850586,
"kl": 1.218994140625,
"learning_rate": 1e-06,
"loss": -0.085,
"num_tokens": 10403062.0,
"reward": 9.69482421875,
"reward_std": 10.32385540008545,
"rewards/rm_reward_func/mean": 9.69482421875,
"rewards/rm_reward_func/std": 19.279569625854492,
"step": 782
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 488.0,
"completions/mean_length": 271.0625,
"completions/mean_terminated_length": 236.6428680419922,
"completions/min_length": 77.0,
"completions/min_terminated_length": 77.0,
"epoch": 0.6264,
"grad_norm": 13.157743453979492,
"kl": 1.40673828125,
"learning_rate": 1e-06,
"loss": 0.0223,
"num_tokens": 10414368.0,
"reward": 5.6474609375,
"reward_std": 5.8091206550598145,
"rewards/rm_reward_func/mean": 5.6474609375,
"rewards/rm_reward_func/std": 13.779108047485352,
"step": 783
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 496.0,
"completions/mean_length": 284.125,
"completions/mean_terminated_length": 268.933349609375,
"completions/min_length": 77.0,
"completions/min_terminated_length": 77.0,
"epoch": 0.6272,
"grad_norm": 4.267988681793213,
"kl": 0.6849365234375,
"learning_rate": 1e-06,
"loss": 0.0464,
"num_tokens": 10426644.0,
"reward": 8.6116943359375,
"reward_std": 8.114376068115234,
"rewards/rm_reward_func/mean": 8.6116943359375,
"rewards/rm_reward_func/std": 13.790817260742188,
"step": 784
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 501.0,
"completions/mean_length": 310.46875,
"completions/mean_terminated_length": 231.60870361328125,
"completions/min_length": 76.0,
"completions/min_terminated_length": 76.0,
"epoch": 0.628,
"grad_norm": 26.008434295654297,
"kl": 1.810791015625,
"learning_rate": 1e-06,
"loss": 0.0316,
"num_tokens": 10443571.0,
"reward": 7.016845703125,
"reward_std": 8.47242259979248,
"rewards/rm_reward_func/mean": 7.016845703125,
"rewards/rm_reward_func/std": 13.28702449798584,
"step": 785
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 498.0,
"completions/mean_length": 244.15625,
"completions/mean_terminated_length": 216.44827270507812,
"completions/min_length": 63.0,
"completions/min_terminated_length": 63.0,
"epoch": 0.6288,
"grad_norm": 18.478044509887695,
"kl": 6.23046875,
"learning_rate": 1e-06,
"loss": 0.1698,
"num_tokens": 10453696.0,
"reward": 2.2125244140625,
"reward_std": 8.634559631347656,
"rewards/rm_reward_func/mean": 2.2125244140625,
"rewards/rm_reward_func/std": 14.19184684753418,
"step": 786
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 463.0,
"completions/mean_length": 280.875,
"completions/mean_terminated_length": 238.07408142089844,
"completions/min_length": 42.0,
"completions/min_terminated_length": 42.0,
"epoch": 0.6296,
"grad_norm": 8.919168472290039,
"kl": 0.99560546875,
"learning_rate": 1e-06,
"loss": -0.0556,
"num_tokens": 10467428.0,
"reward": -6.05810546875,
"reward_std": 5.291695594787598,
"rewards/rm_reward_func/mean": -6.05810546875,
"rewards/rm_reward_func/std": 14.077133178710938,
"step": 787
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 350.96875,
"completions/mean_terminated_length": 297.29168701171875,
"completions/min_length": 62.0,
"completions/min_terminated_length": 62.0,
"epoch": 0.6304,
"grad_norm": 10.0316743850708,
"kl": 4.890625,
"learning_rate": 1e-06,
"loss": -0.1151,
"num_tokens": 10480643.0,
"reward": -2.29998779296875,
"reward_std": 18.708683013916016,
"rewards/rm_reward_func/mean": -2.29998779296875,
"rewards/rm_reward_func/std": 21.16217041015625,
"step": 788
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 479.0,
"completions/mean_length": 298.5,
"completions/mean_terminated_length": 284.2666931152344,
"completions/min_length": 16.0,
"completions/min_terminated_length": 16.0,
"epoch": 0.6312,
"grad_norm": 21.06656837463379,
"kl": 6.18212890625,
"learning_rate": 1e-06,
"loss": 0.3144,
"num_tokens": 10492979.0,
"reward": -6.215576171875,
"reward_std": 7.852952003479004,
"rewards/rm_reward_func/mean": -6.215576171875,
"rewards/rm_reward_func/std": 14.924491882324219,
"step": 789
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 495.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 184.15625,
"completions/mean_terminated_length": 184.15625,
"completions/min_length": 60.0,
"completions/min_terminated_length": 60.0,
"epoch": 0.632,
"grad_norm": 6.202207565307617,
"kl": 1.20654296875,
"learning_rate": 1e-06,
"loss": 0.0211,
"num_tokens": 10503632.0,
"reward": 7.91845703125,
"reward_std": 6.895031452178955,
"rewards/rm_reward_func/mean": 7.91845703125,
"rewards/rm_reward_func/std": 13.781877517700195,
"step": 790
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 505.0,
"completions/mean_length": 200.8125,
"completions/mean_terminated_length": 168.6206817626953,
"completions/min_length": 58.0,
"completions/min_terminated_length": 58.0,
"epoch": 0.6328,
"grad_norm": 19.258525848388672,
"kl": 2.4609375,
"learning_rate": 1e-06,
"loss": -0.188,
"num_tokens": 10512882.0,
"reward": 0.01904296875,
"reward_std": 13.03593635559082,
"rewards/rm_reward_func/mean": 0.01904296875,
"rewards/rm_reward_func/std": 16.95264434814453,
"step": 791
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 376.71875,
"completions/mean_terminated_length": 331.625,
"completions/min_length": 130.0,
"completions/min_terminated_length": 130.0,
"epoch": 0.6336,
"grad_norm": 15.64734172821045,
"kl": 5.783203125,
"learning_rate": 1e-06,
"loss": 0.1589,
"num_tokens": 10527289.0,
"reward": 0.9697265625,
"reward_std": 10.092090606689453,
"rewards/rm_reward_func/mean": 0.9697265625,
"rewards/rm_reward_func/std": 18.045101165771484,
"step": 792
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 511.0,
"completions/mean_length": 442.59375,
"completions/mean_terminated_length": 419.4583435058594,
"completions/min_length": 307.0,
"completions/min_terminated_length": 307.0,
"epoch": 0.6344,
"grad_norm": 5.187870025634766,
"kl": 2.06005859375,
"learning_rate": 1e-06,
"loss": 0.0698,
"num_tokens": 10543972.0,
"reward": 14.57086181640625,
"reward_std": 9.486396789550781,
"rewards/rm_reward_func/mean": 14.57086181640625,
"rewards/rm_reward_func/std": 18.273242950439453,
"step": 793
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 496.0,
"completions/mean_length": 387.5625,
"completions/mean_terminated_length": 364.5185241699219,
"completions/min_length": 173.0,
"completions/min_terminated_length": 173.0,
"epoch": 0.6352,
"grad_norm": 14.29277229309082,
"kl": 4.53125,
"learning_rate": 1e-06,
"loss": 0.1114,
"num_tokens": 10560582.0,
"reward": -0.26617431640625,
"reward_std": 8.553365707397461,
"rewards/rm_reward_func/mean": -0.26617431640625,
"rewards/rm_reward_func/std": 10.436263084411621,
"step": 794
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 236.59375,
"completions/mean_terminated_length": 227.7096710205078,
"completions/min_length": 77.0,
"completions/min_terminated_length": 77.0,
"epoch": 0.636,
"grad_norm": 7.5032501220703125,
"kl": 1.154296875,
"learning_rate": 1e-06,
"loss": 0.0065,
"num_tokens": 10571985.0,
"reward": 7.23681640625,
"reward_std": 8.498395919799805,
"rewards/rm_reward_func/mean": 7.23681640625,
"rewards/rm_reward_func/std": 13.58292007446289,
"step": 795
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 448.0,
"completions/mean_length": 239.15625,
"completions/mean_terminated_length": 162.75999450683594,
"completions/min_length": 54.0,
"completions/min_terminated_length": 54.0,
"epoch": 0.6368,
"grad_norm": 21.288700103759766,
"kl": 6.2919921875,
"learning_rate": 1e-06,
"loss": 0.1455,
"num_tokens": 10582014.0,
"reward": -8.60565185546875,
"reward_std": 5.36991548538208,
"rewards/rm_reward_func/mean": -8.60565185546875,
"rewards/rm_reward_func/std": 15.003857612609863,
"step": 796
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 247.03125,
"completions/mean_terminated_length": 209.17857360839844,
"completions/min_length": 76.0,
"completions/min_terminated_length": 76.0,
"epoch": 0.6376,
"grad_norm": 7.441081523895264,
"kl": 2.571533203125,
"learning_rate": 1e-06,
"loss": 0.0626,
"num_tokens": 10592679.0,
"reward": 2.77825927734375,
"reward_std": 2.551882028579712,
"rewards/rm_reward_func/mean": 2.77825927734375,
"rewards/rm_reward_func/std": 8.97035026550293,
"step": 797
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 487.0,
"completions/mean_length": 176.28125,
"completions/mean_terminated_length": 165.4516143798828,
"completions/min_length": 66.0,
"completions/min_terminated_length": 66.0,
"epoch": 0.6384,
"grad_norm": 6.360789775848389,
"kl": 0.4638671875,
"learning_rate": 1e-06,
"loss": 0.0344,
"num_tokens": 10604672.0,
"reward": 15.427978515625,
"reward_std": 2.812357187271118,
"rewards/rm_reward_func/mean": 15.427978515625,
"rewards/rm_reward_func/std": 14.900362014770508,
"step": 798
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 476.0,
"completions/mean_length": 338.125,
"completions/mean_terminated_length": 259.0909118652344,
"completions/min_length": 81.0,
"completions/min_terminated_length": 81.0,
"epoch": 0.6392,
"grad_norm": 6.797297954559326,
"kl": 0.995361328125,
"learning_rate": 1e-06,
"loss": 0.0529,
"num_tokens": 10621628.0,
"reward": 0.404541015625,
"reward_std": 7.615126609802246,
"rewards/rm_reward_func/mean": 0.404541015625,
"rewards/rm_reward_func/std": 12.74864673614502,
"step": 799
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 494.0,
"completions/mean_length": 421.46875,
"completions/mean_terminated_length": 374.0476379394531,
"completions/min_length": 155.0,
"completions/min_terminated_length": 155.0,
"epoch": 0.64,
"grad_norm": 9.847874641418457,
"kl": 2.7900390625,
"learning_rate": 1e-06,
"loss": 0.0953,
"num_tokens": 10638051.0,
"reward": 3.163330078125,
"reward_std": 10.612391471862793,
"rewards/rm_reward_func/mean": 3.163330078125,
"rewards/rm_reward_func/std": 15.437744140625,
"step": 800
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 455.0,
"completions/mean_length": 216.84375,
"completions/mean_terminated_length": 148.73077392578125,
"completions/min_length": 39.0,
"completions/min_terminated_length": 39.0,
"epoch": 0.6408,
"grad_norm": 10.988313674926758,
"kl": 3.015625,
"learning_rate": 1e-06,
"loss": 0.1434,
"num_tokens": 10647630.0,
"reward": -3.7789306640625,
"reward_std": 6.626611232757568,
"rewards/rm_reward_func/mean": -3.7789306640625,
"rewards/rm_reward_func/std": 10.867609977722168,
"step": 801
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 401.0625,
"completions/mean_terminated_length": 364.0833435058594,
"completions/min_length": 97.0,
"completions/min_terminated_length": 97.0,
"epoch": 0.6416,
"grad_norm": 6.588342666625977,
"kl": 1.3544921875,
"learning_rate": 1e-06,
"loss": -0.0717,
"num_tokens": 10665272.0,
"reward": 3.876220703125,
"reward_std": 8.485553741455078,
"rewards/rm_reward_func/mean": 3.876220703125,
"rewards/rm_reward_func/std": 11.902339935302734,
"step": 802
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 498.0,
"completions/mean_length": 264.9375,
"completions/mean_terminated_length": 256.9677429199219,
"completions/min_length": 41.0,
"completions/min_terminated_length": 41.0,
"epoch": 0.6424,
"grad_norm": 16.689680099487305,
"kl": 4.962890625,
"learning_rate": 1e-06,
"loss": 0.2397,
"num_tokens": 10676294.0,
"reward": -3.4169921875,
"reward_std": 6.7452592849731445,
"rewards/rm_reward_func/mean": -3.4169921875,
"rewards/rm_reward_func/std": 19.948022842407227,
"step": 803
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 485.0,
"completions/max_terminated_length": 485.0,
"completions/mean_length": 270.46875,
"completions/mean_terminated_length": 270.46875,
"completions/min_length": 79.0,
"completions/min_terminated_length": 79.0,
"epoch": 0.6432,
"grad_norm": 6.953851222991943,
"kl": 1.05712890625,
"learning_rate": 1e-06,
"loss": -0.0594,
"num_tokens": 10690029.0,
"reward": -0.96844482421875,
"reward_std": 4.365692138671875,
"rewards/rm_reward_func/mean": -0.96844482421875,
"rewards/rm_reward_func/std": 6.488677024841309,
"step": 804
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 277.65625,
"completions/mean_terminated_length": 270.0967712402344,
"completions/min_length": 78.0,
"completions/min_terminated_length": 78.0,
"epoch": 0.644,
"grad_norm": 6.296433448791504,
"kl": 1.06787109375,
"learning_rate": 1e-06,
"loss": -0.0623,
"num_tokens": 10701498.0,
"reward": 1.8621826171875,
"reward_std": 4.512002468109131,
"rewards/rm_reward_func/mean": 1.8621826171875,
"rewards/rm_reward_func/std": 13.205521583557129,
"step": 805
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 211.0625,
"completions/mean_terminated_length": 201.35482788085938,
"completions/min_length": 78.0,
"completions/min_terminated_length": 78.0,
"epoch": 0.6448,
"grad_norm": 6.924914836883545,
"kl": 2.736328125,
"learning_rate": 1e-06,
"loss": 0.0839,
"num_tokens": 10712828.0,
"reward": 4.1123046875,
"reward_std": 2.182391405105591,
"rewards/rm_reward_func/mean": 4.1123046875,
"rewards/rm_reward_func/std": 6.567470073699951,
"step": 806
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 496.0,
"completions/mean_length": 256.65625,
"completions/mean_terminated_length": 209.37037658691406,
"completions/min_length": 47.0,
"completions/min_terminated_length": 47.0,
"epoch": 0.6456,
"grad_norm": 18.669151306152344,
"kl": 6.09765625,
"learning_rate": 1e-06,
"loss": 0.2974,
"num_tokens": 10724233.0,
"reward": -15.416748046875,
"reward_std": 7.353342056274414,
"rewards/rm_reward_func/mean": -15.416748046875,
"rewards/rm_reward_func/std": 11.860098838806152,
"step": 807
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 482.0,
"completions/mean_length": 396.25,
"completions/mean_terminated_length": 326.8000183105469,
"completions/min_length": 140.0,
"completions/min_terminated_length": 140.0,
"epoch": 0.6464,
"grad_norm": 6.7639875411987305,
"kl": 4.123046875,
"learning_rate": 1e-06,
"loss": 0.1232,
"num_tokens": 10739529.0,
"reward": -0.310546875,
"reward_std": 9.28040885925293,
"rewards/rm_reward_func/mean": -0.310546875,
"rewards/rm_reward_func/std": 11.789945602416992,
"step": 808
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 505.0,
"completions/mean_length": 353.40625,
"completions/mean_terminated_length": 324.03704833984375,
"completions/min_length": 136.0,
"completions/min_terminated_length": 136.0,
"epoch": 0.6472,
"grad_norm": 11.955965042114258,
"kl": 5.74609375,
"learning_rate": 1e-06,
"loss": 0.123,
"num_tokens": 10756438.0,
"reward": 1.25439453125,
"reward_std": 12.41511058807373,
"rewards/rm_reward_func/mean": 1.25439453125,
"rewards/rm_reward_func/std": 13.135236740112305,
"step": 809
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 475.0,
"completions/mean_length": 262.03125,
"completions/mean_terminated_length": 236.1724090576172,
"completions/min_length": 69.0,
"completions/min_terminated_length": 69.0,
"epoch": 0.648,
"grad_norm": 6.4916911125183105,
"kl": 4.392578125,
"learning_rate": 1e-06,
"loss": 0.1702,
"num_tokens": 10767359.0,
"reward": -5.23974609375,
"reward_std": 5.122437477111816,
"rewards/rm_reward_func/mean": -5.23974609375,
"rewards/rm_reward_func/std": 10.06690502166748,
"step": 810
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 358.96875,
"completions/mean_terminated_length": 330.629638671875,
"completions/min_length": 110.0,
"completions/min_terminated_length": 110.0,
"epoch": 0.6488,
"grad_norm": 38.71643829345703,
"kl": 10.5546875,
"learning_rate": 1e-06,
"loss": 0.3489,
"num_tokens": 10780798.0,
"reward": -9.0185546875,
"reward_std": 5.921327590942383,
"rewards/rm_reward_func/mean": -9.0185546875,
"rewards/rm_reward_func/std": 9.659137725830078,
"step": 811
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 511.0,
"completions/mean_length": 368.40625,
"completions/mean_terminated_length": 358.8333435058594,
"completions/min_length": 65.0,
"completions/min_terminated_length": 65.0,
"epoch": 0.6496,
"grad_norm": 8.951486587524414,
"kl": 3.515625,
"learning_rate": 1e-06,
"loss": -0.0127,
"num_tokens": 10795059.0,
"reward": -1.08935546875,
"reward_std": 11.639703750610352,
"rewards/rm_reward_func/mean": -1.08935546875,
"rewards/rm_reward_func/std": 14.727822303771973,
"step": 812
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 480.0,
"completions/mean_length": 289.21875,
"completions/mean_terminated_length": 237.80770874023438,
"completions/min_length": 46.0,
"completions/min_terminated_length": 46.0,
"epoch": 0.6504,
"grad_norm": 21.51280403137207,
"kl": 4.89453125,
"learning_rate": 1e-06,
"loss": 0.0933,
"num_tokens": 10806426.0,
"reward": -5.33056640625,
"reward_std": 8.251094818115234,
"rewards/rm_reward_func/mean": -5.33056640625,
"rewards/rm_reward_func/std": 14.58669376373291,
"step": 813
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 485.0,
"completions/mean_length": 325.90625,
"completions/mean_terminated_length": 273.79998779296875,
"completions/min_length": 79.0,
"completions/min_terminated_length": 79.0,
"epoch": 0.6512,
"grad_norm": 14.707463264465332,
"kl": 6.0205078125,
"learning_rate": 1e-06,
"loss": 0.1921,
"num_tokens": 10820431.0,
"reward": -2.914306640625,
"reward_std": 8.024243354797363,
"rewards/rm_reward_func/mean": -2.914306640625,
"rewards/rm_reward_func/std": 14.138604164123535,
"step": 814
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 504.0,
"completions/mean_length": 264.0625,
"completions/mean_terminated_length": 256.06451416015625,
"completions/min_length": 52.0,
"completions/min_terminated_length": 52.0,
"epoch": 0.652,
"grad_norm": 9.099730491638184,
"kl": 2.73681640625,
"learning_rate": 1e-06,
"loss": 0.0878,
"num_tokens": 10831561.0,
"reward": -2.70703125,
"reward_std": 6.87619686126709,
"rewards/rm_reward_func/mean": -2.70703125,
"rewards/rm_reward_func/std": 13.68471908569336,
"step": 815
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 276.15625,
"completions/mean_terminated_length": 260.433349609375,
"completions/min_length": 77.0,
"completions/min_terminated_length": 77.0,
"epoch": 0.6528,
"grad_norm": 9.196378707885742,
"kl": 4.783203125,
"learning_rate": 1e-06,
"loss": 0.0501,
"num_tokens": 10843046.0,
"reward": -3.22216796875,
"reward_std": 6.319431781768799,
"rewards/rm_reward_func/mean": -3.22216796875,
"rewards/rm_reward_func/std": 8.792978286743164,
"step": 816
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 403.0,
"completions/max_terminated_length": 403.0,
"completions/mean_length": 156.75,
"completions/mean_terminated_length": 156.75,
"completions/min_length": 30.0,
"completions/min_terminated_length": 30.0,
"epoch": 0.6536,
"grad_norm": 7.186251163482666,
"kl": 5.7412109375,
"learning_rate": 1e-06,
"loss": 0.248,
"num_tokens": 10851550.0,
"reward": 1.57763671875,
"reward_std": 8.453566551208496,
"rewards/rm_reward_func/mean": 1.57763671875,
"rewards/rm_reward_func/std": 19.387672424316406,
"step": 817
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 292.71875,
"completions/mean_terminated_length": 270.03448486328125,
"completions/min_length": 73.0,
"completions/min_terminated_length": 73.0,
"epoch": 0.6544,
"grad_norm": 9.593155860900879,
"kl": 1.69677734375,
"learning_rate": 1e-06,
"loss": 0.0271,
"num_tokens": 10865029.0,
"reward": 5.674560546875,
"reward_std": 6.335400581359863,
"rewards/rm_reward_func/mean": 5.674560546875,
"rewards/rm_reward_func/std": 13.289620399475098,
"step": 818
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 442.0,
"completions/mean_length": 311.0,
"completions/mean_terminated_length": 244.0,
"completions/min_length": 75.0,
"completions/min_terminated_length": 75.0,
"epoch": 0.6552,
"grad_norm": 9.963946342468262,
"kl": 1.982421875,
"learning_rate": 1e-06,
"loss": 0.0487,
"num_tokens": 10877253.0,
"reward": -0.123199462890625,
"reward_std": 6.386770248413086,
"rewards/rm_reward_func/mean": -0.123199462890625,
"rewards/rm_reward_func/std": 10.461974143981934,
"step": 819
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 426.0,
"completions/mean_length": 358.9375,
"completions/mean_terminated_length": 307.91668701171875,
"completions/min_length": 156.0,
"completions/min_terminated_length": 156.0,
"epoch": 0.656,
"grad_norm": 6.363739013671875,
"kl": 2.058837890625,
"learning_rate": 1e-06,
"loss": 0.1152,
"num_tokens": 10893083.0,
"reward": -0.5570068359375,
"reward_std": 6.5521240234375,
"rewards/rm_reward_func/mean": -0.5570068359375,
"rewards/rm_reward_func/std": 18.305418014526367,
"step": 820
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 319.75,
"completions/mean_terminated_length": 299.862060546875,
"completions/min_length": 151.0,
"completions/min_terminated_length": 151.0,
"epoch": 0.6568,
"grad_norm": 8.368305206298828,
"kl": 3.38720703125,
"learning_rate": 1e-06,
"loss": -0.0173,
"num_tokens": 10906779.0,
"reward": 4.049560546875,
"reward_std": 14.061559677124023,
"rewards/rm_reward_func/mean": 4.049560546875,
"rewards/rm_reward_func/std": 21.243824005126953,
"step": 821
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 486.0,
"completions/mean_length": 324.6875,
"completions/mean_terminated_length": 290.0,
"completions/min_length": 53.0,
"completions/min_terminated_length": 53.0,
"epoch": 0.6576,
"grad_norm": 8.28355598449707,
"kl": 2.156005859375,
"learning_rate": 1e-06,
"loss": -0.1195,
"num_tokens": 10919833.0,
"reward": -12.810791015625,
"reward_std": 5.67118501663208,
"rewards/rm_reward_func/mean": -12.810791015625,
"rewards/rm_reward_func/std": 7.540193557739258,
"step": 822
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 424.0,
"completions/max_terminated_length": 424.0,
"completions/mean_length": 167.9375,
"completions/mean_terminated_length": 167.9375,
"completions/min_length": 19.0,
"completions/min_terminated_length": 19.0,
"epoch": 0.6584,
"grad_norm": 7.692931175231934,
"kl": 4.6162109375,
"learning_rate": 1e-06,
"loss": 0.2633,
"num_tokens": 10932023.0,
"reward": -4.395263671875,
"reward_std": 5.285364151000977,
"rewards/rm_reward_func/mean": -4.395263671875,
"rewards/rm_reward_func/std": 9.415786743164062,
"step": 823
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 262.84375,
"completions/mean_terminated_length": 193.0800018310547,
"completions/min_length": 30.0,
"completions/min_terminated_length": 30.0,
"epoch": 0.6592,
"grad_norm": 11.74027156829834,
"kl": 0.53271484375,
"learning_rate": 1e-06,
"loss": -0.1976,
"num_tokens": 10943026.0,
"reward": 4.1466064453125,
"reward_std": 7.172249794006348,
"rewards/rm_reward_func/mean": 4.1466064453125,
"rewards/rm_reward_func/std": 11.245109558105469,
"step": 824
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 490.0,
"completions/mean_length": 314.4375,
"completions/mean_terminated_length": 237.13043212890625,
"completions/min_length": 81.0,
"completions/min_terminated_length": 81.0,
"epoch": 0.66,
"grad_norm": 4.412758827209473,
"kl": 0.615966796875,
"learning_rate": 1e-06,
"loss": 0.0277,
"num_tokens": 10957904.0,
"reward": -2.131103515625,
"reward_std": 4.323854446411133,
"rewards/rm_reward_func/mean": -2.131103515625,
"rewards/rm_reward_func/std": 8.868707656860352,
"step": 825
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 462.0,
"completions/mean_length": 349.3125,
"completions/mean_terminated_length": 319.1851806640625,
"completions/min_length": 119.0,
"completions/min_terminated_length": 119.0,
"epoch": 0.6608,
"grad_norm": 8.348759651184082,
"kl": 1.59619140625,
"learning_rate": 1e-06,
"loss": 0.065,
"num_tokens": 10971842.0,
"reward": 0.430908203125,
"reward_std": 5.89361047744751,
"rewards/rm_reward_func/mean": 0.430908203125,
"rewards/rm_reward_func/std": 11.598136901855469,
"step": 826
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 478.0,
"completions/max_terminated_length": 478.0,
"completions/mean_length": 261.21875,
"completions/mean_terminated_length": 261.21875,
"completions/min_length": 74.0,
"completions/min_terminated_length": 74.0,
"epoch": 0.6616,
"grad_norm": 10.604769706726074,
"kl": 1.353515625,
"learning_rate": 1e-06,
"loss": -0.1187,
"num_tokens": 10982777.0,
"reward": 1.006195068359375,
"reward_std": 13.28079891204834,
"rewards/rm_reward_func/mean": 1.006195068359375,
"rewards/rm_reward_func/std": 14.09341049194336,
"step": 827
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 511.0,
"completions/mean_length": 333.65625,
"completions/mean_terminated_length": 321.7666931152344,
"completions/min_length": 58.0,
"completions/min_terminated_length": 58.0,
"epoch": 0.6624,
"grad_norm": 6.49513053894043,
"kl": 2.354248046875,
"learning_rate": 1e-06,
"loss": 0.1536,
"num_tokens": 10996190.0,
"reward": 4.603271484375,
"reward_std": 7.740581035614014,
"rewards/rm_reward_func/mean": 4.603271484375,
"rewards/rm_reward_func/std": 18.76112937927246,
"step": 828
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 470.0,
"completions/mean_length": 340.15625,
"completions/mean_terminated_length": 282.875,
"completions/min_length": 30.0,
"completions/min_terminated_length": 30.0,
"epoch": 0.6632,
"grad_norm": 9.52800464630127,
"kl": 2.73828125,
"learning_rate": 1e-06,
"loss": -0.0569,
"num_tokens": 11009467.0,
"reward": 3.36962890625,
"reward_std": 10.40433120727539,
"rewards/rm_reward_func/mean": 3.36962890625,
"rewards/rm_reward_func/std": 13.418220520019531,
"step": 829
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 498.0,
"completions/mean_length": 240.65625,
"completions/mean_terminated_length": 222.56668090820312,
"completions/min_length": 81.0,
"completions/min_terminated_length": 81.0,
"epoch": 0.664,
"grad_norm": 10.561025619506836,
"kl": 2.5478515625,
"learning_rate": 1e-06,
"loss": 0.0396,
"num_tokens": 11020232.0,
"reward": 8.3240966796875,
"reward_std": 10.61776351928711,
"rewards/rm_reward_func/mean": 8.3240966796875,
"rewards/rm_reward_func/std": 14.970850944519043,
"step": 830
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 474.0,
"completions/mean_length": 330.1875,
"completions/mean_terminated_length": 279.2799987792969,
"completions/min_length": 73.0,
"completions/min_terminated_length": 73.0,
"epoch": 0.6648,
"grad_norm": 11.870400428771973,
"kl": 3.390625,
"learning_rate": 1e-06,
"loss": 0.0118,
"num_tokens": 11033358.0,
"reward": 5.705322265625,
"reward_std": 7.648243427276611,
"rewards/rm_reward_func/mean": 5.705322265625,
"rewards/rm_reward_func/std": 19.251300811767578,
"step": 831
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 503.0,
"completions/mean_length": 383.34375,
"completions/mean_terminated_length": 370.03448486328125,
"completions/min_length": 145.0,
"completions/min_terminated_length": 145.0,
"epoch": 0.6656,
"grad_norm": 7.473958969116211,
"kl": 0.41357421875,
"learning_rate": 1e-06,
"loss": 0.0041,
"num_tokens": 11050633.0,
"reward": 9.1138916015625,
"reward_std": 4.864926338195801,
"rewards/rm_reward_func/mean": 9.1138916015625,
"rewards/rm_reward_func/std": 15.038235664367676,
"step": 832
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 295.90625,
"completions/mean_terminated_length": 288.93548583984375,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.6664,
"grad_norm": 8.113051414489746,
"kl": 1.4638671875,
"learning_rate": 1e-06,
"loss": 0.0144,
"num_tokens": 11063022.0,
"reward": 3.6669921875,
"reward_std": 7.224357604980469,
"rewards/rm_reward_func/mean": 3.6669921875,
"rewards/rm_reward_func/std": 12.741311073303223,
"step": 833
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 489.0,
"completions/mean_length": 336.53125,
"completions/mean_terminated_length": 330.8709716796875,
"completions/min_length": 180.0,
"completions/min_terminated_length": 180.0,
"epoch": 0.6672,
"grad_norm": 6.7608513832092285,
"kl": 0.941162109375,
"learning_rate": 1e-06,
"loss": -0.0642,
"num_tokens": 11076839.0,
"reward": 0.627685546875,
"reward_std": 4.866849422454834,
"rewards/rm_reward_func/mean": 0.627685546875,
"rewards/rm_reward_func/std": 9.038817405700684,
"step": 834
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 511.0,
"completions/mean_length": 283.34375,
"completions/mean_terminated_length": 268.1000061035156,
"completions/min_length": 52.0,
"completions/min_terminated_length": 52.0,
"epoch": 0.668,
"grad_norm": 22.115230560302734,
"kl": 7.388671875,
"learning_rate": 1e-06,
"loss": 0.1979,
"num_tokens": 11091042.0,
"reward": 0.4773712158203125,
"reward_std": 9.4666109085083,
"rewards/rm_reward_func/mean": 0.4773712158203125,
"rewards/rm_reward_func/std": 18.32492446899414,
"step": 835
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 397.0,
"completions/mean_length": 184.3125,
"completions/mean_terminated_length": 162.4666748046875,
"completions/min_length": 29.0,
"completions/min_terminated_length": 29.0,
"epoch": 0.6688,
"grad_norm": 31.762920379638672,
"kl": 12.359375,
"learning_rate": 1e-06,
"loss": 0.5706,
"num_tokens": 11102388.0,
"reward": -8.64892578125,
"reward_std": 7.599771499633789,
"rewards/rm_reward_func/mean": -8.64892578125,
"rewards/rm_reward_func/std": 9.485981941223145,
"step": 836
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 424.0,
"completions/mean_length": 242.21875,
"completions/mean_terminated_length": 192.25926208496094,
"completions/min_length": 46.0,
"completions/min_terminated_length": 46.0,
"epoch": 0.6696,
"grad_norm": 27.026330947875977,
"kl": 9.296875,
"learning_rate": 1e-06,
"loss": 0.2068,
"num_tokens": 11115539.0,
"reward": -7.0478515625,
"reward_std": 10.2100248336792,
"rewards/rm_reward_func/mean": -7.0478515625,
"rewards/rm_reward_func/std": 18.10556411743164,
"step": 837
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 494.0,
"completions/max_terminated_length": 494.0,
"completions/mean_length": 272.75,
"completions/mean_terminated_length": 272.75,
"completions/min_length": 55.0,
"completions/min_terminated_length": 55.0,
"epoch": 0.6704,
"grad_norm": 8.460334777832031,
"kl": 2.99609375,
"learning_rate": 1e-06,
"loss": -0.0436,
"num_tokens": 11126803.0,
"reward": 3.97998046875,
"reward_std": 8.513212203979492,
"rewards/rm_reward_func/mean": 3.97998046875,
"rewards/rm_reward_func/std": 12.565332412719727,
"step": 838
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 388.0,
"completions/max_terminated_length": 388.0,
"completions/mean_length": 158.90625,
"completions/mean_terminated_length": 158.90625,
"completions/min_length": 77.0,
"completions/min_terminated_length": 77.0,
"epoch": 0.6712,
"grad_norm": 7.5811028480529785,
"kl": 0.560546875,
"learning_rate": 1e-06,
"loss": 0.0078,
"num_tokens": 11135056.0,
"reward": 10.41888427734375,
"reward_std": 2.0775063037872314,
"rewards/rm_reward_func/mean": 10.41888427734375,
"rewards/rm_reward_func/std": 4.920321941375732,
"step": 839
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 481.0,
"completions/mean_length": 194.75,
"completions/mean_terminated_length": 173.60000610351562,
"completions/min_length": 97.0,
"completions/min_terminated_length": 97.0,
"epoch": 0.672,
"grad_norm": 17.789752960205078,
"kl": 7.279296875,
"learning_rate": 1e-06,
"loss": 0.2592,
"num_tokens": 11143560.0,
"reward": -6.46221923828125,
"reward_std": 8.128403663635254,
"rewards/rm_reward_func/mean": -6.46221923828125,
"rewards/rm_reward_func/std": 8.697508811950684,
"step": 840
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 324.78125,
"completions/mean_terminated_length": 318.7419128417969,
"completions/min_length": 77.0,
"completions/min_terminated_length": 77.0,
"epoch": 0.6728,
"grad_norm": 7.485666751861572,
"kl": 1.71435546875,
"learning_rate": 1e-06,
"loss": 0.0202,
"num_tokens": 11158129.0,
"reward": 14.0751953125,
"reward_std": 8.069329261779785,
"rewards/rm_reward_func/mean": 14.0751953125,
"rewards/rm_reward_func/std": 10.917937278747559,
"step": 841
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 448.0,
"completions/max_terminated_length": 448.0,
"completions/mean_length": 282.53125,
"completions/mean_terminated_length": 282.53125,
"completions/min_length": 78.0,
"completions/min_terminated_length": 78.0,
"epoch": 0.6736,
"grad_norm": 8.429314613342285,
"kl": 3.7744140625,
"learning_rate": 1e-06,
"loss": 0.1521,
"num_tokens": 11170770.0,
"reward": 8.86328125,
"reward_std": 12.041173934936523,
"rewards/rm_reward_func/mean": 8.86328125,
"rewards/rm_reward_func/std": 16.136621475219727,
"step": 842
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 335.0,
"completions/mean_length": 147.4375,
"completions/mean_terminated_length": 109.72413635253906,
"completions/min_length": 73.0,
"completions/min_terminated_length": 73.0,
"epoch": 0.6744,
"grad_norm": 7.667215347290039,
"kl": 2.34375,
"learning_rate": 1e-06,
"loss": 0.0167,
"num_tokens": 11182616.0,
"reward": 9.590576171875,
"reward_std": 2.2192811965942383,
"rewards/rm_reward_func/mean": 9.590576171875,
"rewards/rm_reward_func/std": 11.049510955810547,
"step": 843
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 389.0,
"completions/mean_length": 146.96875,
"completions/mean_terminated_length": 135.19354248046875,
"completions/min_length": 32.0,
"completions/min_terminated_length": 32.0,
"epoch": 0.6752,
"grad_norm": 28.290157318115234,
"kl": 6.166015625,
"learning_rate": 1e-06,
"loss": 0.2478,
"num_tokens": 11192607.0,
"reward": -7.3682861328125,
"reward_std": 4.421894073486328,
"rewards/rm_reward_func/mean": -7.3682861328125,
"rewards/rm_reward_func/std": 13.098028182983398,
"step": 844
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 437.0,
"completions/mean_length": 278.875,
"completions/mean_terminated_length": 271.3548278808594,
"completions/min_length": 147.0,
"completions/min_terminated_length": 147.0,
"epoch": 0.676,
"grad_norm": 18.575132369995117,
"kl": 5.462890625,
"learning_rate": 1e-06,
"loss": 0.1575,
"num_tokens": 11204379.0,
"reward": -2.055908203125,
"reward_std": 6.175776481628418,
"rewards/rm_reward_func/mean": -2.055908203125,
"rewards/rm_reward_func/std": 10.646092414855957,
"step": 845
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 335.0,
"completions/max_terminated_length": 335.0,
"completions/mean_length": 123.96875,
"completions/mean_terminated_length": 123.96875,
"completions/min_length": 52.0,
"completions/min_terminated_length": 52.0,
"epoch": 0.6768,
"grad_norm": 14.692407608032227,
"kl": 4.3916015625,
"learning_rate": 1e-06,
"loss": 0.1037,
"num_tokens": 11213498.0,
"reward": 2.0,
"reward_std": 3.80605149269104,
"rewards/rm_reward_func/mean": 2.0,
"rewards/rm_reward_func/std": 9.591236114501953,
"step": 846
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 504.0,
"completions/mean_length": 285.75,
"completions/mean_terminated_length": 278.45159912109375,
"completions/min_length": 48.0,
"completions/min_terminated_length": 48.0,
"epoch": 0.6776,
"grad_norm": 9.11420726776123,
"kl": 1.35791015625,
"learning_rate": 1e-06,
"loss": -0.1391,
"num_tokens": 11224962.0,
"reward": 7.869140625,
"reward_std": 11.650513648986816,
"rewards/rm_reward_func/mean": 7.869140625,
"rewards/rm_reward_func/std": 14.315463066101074,
"step": 847
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 411.0,
"completions/max_terminated_length": 411.0,
"completions/mean_length": 196.5625,
"completions/mean_terminated_length": 196.5625,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.6784,
"grad_norm": 9.939024925231934,
"kl": 0.9248046875,
"learning_rate": 1e-06,
"loss": -0.0194,
"num_tokens": 11233572.0,
"reward": 7.11181640625,
"reward_std": 5.81515645980835,
"rewards/rm_reward_func/mean": 7.11181640625,
"rewards/rm_reward_func/std": 10.113432884216309,
"step": 848
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 374.0,
"completions/max_terminated_length": 374.0,
"completions/mean_length": 218.65625,
"completions/mean_terminated_length": 218.65625,
"completions/min_length": 33.0,
"completions/min_terminated_length": 33.0,
"epoch": 0.6792,
"grad_norm": 14.37784194946289,
"kl": 2.833984375,
"learning_rate": 1e-06,
"loss": -0.0176,
"num_tokens": 11242921.0,
"reward": -6.9840850830078125,
"reward_std": 5.020229816436768,
"rewards/rm_reward_func/mean": -6.9840850830078125,
"rewards/rm_reward_func/std": 8.650106430053711,
"step": 849
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 252.0,
"completions/max_terminated_length": 252.0,
"completions/mean_length": 133.5,
"completions/mean_terminated_length": 133.5,
"completions/min_length": 73.0,
"completions/min_terminated_length": 73.0,
"epoch": 0.68,
"grad_norm": 27.497337341308594,
"kl": 2.4345703125,
"learning_rate": 1e-06,
"loss": 0.0773,
"num_tokens": 11250937.0,
"reward": 6.5244140625,
"reward_std": 7.013378143310547,
"rewards/rm_reward_func/mean": 6.5244140625,
"rewards/rm_reward_func/std": 11.999058723449707,
"step": 850
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 397.40625,
"completions/mean_terminated_length": 381.0357360839844,
"completions/min_length": 279.0,
"completions/min_terminated_length": 279.0,
"epoch": 0.6808,
"grad_norm": 6.854437351226807,
"kl": 0.42431640625,
"learning_rate": 1e-06,
"loss": -0.0064,
"num_tokens": 11266398.0,
"reward": 6.41265869140625,
"reward_std": 3.7932474613189697,
"rewards/rm_reward_func/mean": 6.41265869140625,
"rewards/rm_reward_func/std": 17.261632919311523,
"step": 851
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 500.0,
"completions/max_terminated_length": 500.0,
"completions/mean_length": 158.09375,
"completions/mean_terminated_length": 158.09375,
"completions/min_length": 54.0,
"completions/min_terminated_length": 54.0,
"epoch": 0.6816,
"grad_norm": 10.595882415771484,
"kl": 3.65625,
"learning_rate": 1e-06,
"loss": 0.0449,
"num_tokens": 11274801.0,
"reward": -12.980224609375,
"reward_std": 5.851085662841797,
"rewards/rm_reward_func/mean": -12.980224609375,
"rewards/rm_reward_func/std": 13.411259651184082,
"step": 852
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 232.75,
"completions/mean_terminated_length": 223.74192810058594,
"completions/min_length": 35.0,
"completions/min_terminated_length": 35.0,
"epoch": 0.6824,
"grad_norm": 11.931048393249512,
"kl": 4.76171875,
"learning_rate": 1e-06,
"loss": 0.1431,
"num_tokens": 11284161.0,
"reward": -9.20654296875,
"reward_std": 10.515345573425293,
"rewards/rm_reward_func/mean": -9.20654296875,
"rewards/rm_reward_func/std": 18.860841751098633,
"step": 853
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 253.5625,
"completions/mean_terminated_length": 236.33334350585938,
"completions/min_length": 48.0,
"completions/min_terminated_length": 48.0,
"epoch": 0.6832,
"grad_norm": 13.002371788024902,
"kl": 1.6796875,
"learning_rate": 1e-06,
"loss": 0.0015,
"num_tokens": 11294715.0,
"reward": 5.0428466796875,
"reward_std": 6.735762596130371,
"rewards/rm_reward_func/mean": 5.0428466796875,
"rewards/rm_reward_func/std": 12.883460998535156,
"step": 854
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 482.0,
"completions/mean_length": 225.5625,
"completions/mean_terminated_length": 206.4666748046875,
"completions/min_length": 73.0,
"completions/min_terminated_length": 73.0,
"epoch": 0.684,
"grad_norm": 15.005594253540039,
"kl": 2.9814453125,
"learning_rate": 1e-06,
"loss": -0.0264,
"num_tokens": 11304629.0,
"reward": -3.63232421875,
"reward_std": 7.6953206062316895,
"rewards/rm_reward_func/mean": -3.63232421875,
"rewards/rm_reward_func/std": 14.550228118896484,
"step": 855
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 424.0,
"completions/mean_length": 222.5625,
"completions/mean_terminated_length": 155.7692413330078,
"completions/min_length": 81.0,
"completions/min_terminated_length": 81.0,
"epoch": 0.6848,
"grad_norm": 13.886627197265625,
"kl": 2.037109375,
"learning_rate": 1e-06,
"loss": -0.0007,
"num_tokens": 11317831.0,
"reward": -3.196533203125,
"reward_std": 2.4364099502563477,
"rewards/rm_reward_func/mean": -3.196533203125,
"rewards/rm_reward_func/std": 9.435223579406738,
"step": 856
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 473.0,
"completions/max_terminated_length": 473.0,
"completions/mean_length": 276.1875,
"completions/mean_terminated_length": 276.1875,
"completions/min_length": 35.0,
"completions/min_terminated_length": 35.0,
"epoch": 0.6856,
"grad_norm": 42.20450210571289,
"kl": 2.49951171875,
"learning_rate": 1e-06,
"loss": 0.085,
"num_tokens": 11331389.0,
"reward": -5.261474609375,
"reward_std": 6.263424873352051,
"rewards/rm_reward_func/mean": -5.261474609375,
"rewards/rm_reward_func/std": 8.193347930908203,
"step": 857
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 399.0,
"completions/max_terminated_length": 399.0,
"completions/mean_length": 256.1875,
"completions/mean_terminated_length": 256.1875,
"completions/min_length": 62.0,
"completions/min_terminated_length": 62.0,
"epoch": 0.6864,
"grad_norm": 11.973814010620117,
"kl": 0.7158203125,
"learning_rate": 1e-06,
"loss": 0.0198,
"num_tokens": 11341651.0,
"reward": 8.2965087890625,
"reward_std": 7.567012786865234,
"rewards/rm_reward_func/mean": 8.2965087890625,
"rewards/rm_reward_func/std": 12.142066955566406,
"step": 858
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 487.0,
"completions/mean_length": 231.15625,
"completions/mean_terminated_length": 202.10345458984375,
"completions/min_length": 77.0,
"completions/min_terminated_length": 77.0,
"epoch": 0.6872,
"grad_norm": 5.184216499328613,
"kl": 1.4384765625,
"learning_rate": 1e-06,
"loss": 0.0217,
"num_tokens": 11351872.0,
"reward": 6.35546875,
"reward_std": 2.562025785446167,
"rewards/rm_reward_func/mean": 6.35546875,
"rewards/rm_reward_func/std": 8.541465759277344,
"step": 859
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 454.0,
"completions/max_terminated_length": 454.0,
"completions/mean_length": 188.5625,
"completions/mean_terminated_length": 188.5625,
"completions/min_length": 60.0,
"completions/min_terminated_length": 60.0,
"epoch": 0.688,
"grad_norm": 8.177552223205566,
"kl": 2.0556640625,
"learning_rate": 1e-06,
"loss": 0.0713,
"num_tokens": 11361914.0,
"reward": -1.439453125,
"reward_std": 3.191701889038086,
"rewards/rm_reward_func/mean": -1.439453125,
"rewards/rm_reward_func/std": 13.077706336975098,
"step": 860
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 401.0,
"completions/max_terminated_length": 401.0,
"completions/mean_length": 168.21875,
"completions/mean_terminated_length": 168.21875,
"completions/min_length": 59.0,
"completions/min_terminated_length": 59.0,
"epoch": 0.6888,
"grad_norm": 12.462203979492188,
"kl": 0.489501953125,
"learning_rate": 1e-06,
"loss": -0.0841,
"num_tokens": 11371129.0,
"reward": 0.892578125,
"reward_std": 6.40195369720459,
"rewards/rm_reward_func/mean": 0.892578125,
"rewards/rm_reward_func/std": 15.08119010925293,
"step": 861
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 496.0,
"completions/max_terminated_length": 496.0,
"completions/mean_length": 218.28125,
"completions/mean_terminated_length": 218.28125,
"completions/min_length": 78.0,
"completions/min_terminated_length": 78.0,
"epoch": 0.6896,
"grad_norm": 9.378438949584961,
"kl": 1.27490234375,
"learning_rate": 1e-06,
"loss": 0.0262,
"num_tokens": 11381746.0,
"reward": 8.6956787109375,
"reward_std": 7.514378547668457,
"rewards/rm_reward_func/mean": 8.6956787109375,
"rewards/rm_reward_func/std": 12.007038116455078,
"step": 862
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 452.0,
"completions/max_terminated_length": 452.0,
"completions/mean_length": 251.5,
"completions/mean_terminated_length": 251.5,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.6904,
"grad_norm": 9.017190933227539,
"kl": 0.43896484375,
"learning_rate": 1e-06,
"loss": -0.0491,
"num_tokens": 11395986.0,
"reward": 4.80322265625,
"reward_std": 2.939964771270752,
"rewards/rm_reward_func/mean": 4.80322265625,
"rewards/rm_reward_func/std": 9.644649505615234,
"step": 863
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 502.0,
"completions/mean_length": 288.875,
"completions/mean_terminated_length": 281.6773986816406,
"completions/min_length": 44.0,
"completions/min_terminated_length": 44.0,
"epoch": 0.6912,
"grad_norm": 8.287468910217285,
"kl": 1.0576171875,
"learning_rate": 1e-06,
"loss": -0.0984,
"num_tokens": 11407550.0,
"reward": -3.62890625,
"reward_std": 5.023155689239502,
"rewards/rm_reward_func/mean": -3.62890625,
"rewards/rm_reward_func/std": 15.647568702697754,
"step": 864
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 451.0,
"completions/mean_length": 234.59375,
"completions/mean_terminated_length": 225.64515686035156,
"completions/min_length": 71.0,
"completions/min_terminated_length": 71.0,
"epoch": 0.692,
"grad_norm": 12.379044532775879,
"kl": 1.384765625,
"learning_rate": 1e-06,
"loss": -0.0319,
"num_tokens": 11419385.0,
"reward": 1.34521484375,
"reward_std": 5.130009174346924,
"rewards/rm_reward_func/mean": 1.34521484375,
"rewards/rm_reward_func/std": 16.12451171875,
"step": 865
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 402.0,
"completions/mean_length": 298.53125,
"completions/mean_terminated_length": 249.2692413330078,
"completions/min_length": 94.0,
"completions/min_terminated_length": 94.0,
"epoch": 0.6928,
"grad_norm": 8.24583911895752,
"kl": 2.95263671875,
"learning_rate": 1e-06,
"loss": 0.0369,
"num_tokens": 11434746.0,
"reward": -10.24560546875,
"reward_std": 6.283100128173828,
"rewards/rm_reward_func/mean": -10.24560546875,
"rewards/rm_reward_func/std": 11.58389949798584,
"step": 866
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 402.0,
"completions/max_terminated_length": 402.0,
"completions/mean_length": 245.84375,
"completions/mean_terminated_length": 245.84375,
"completions/min_length": 131.0,
"completions/min_terminated_length": 131.0,
"epoch": 0.6936,
"grad_norm": 11.311857223510742,
"kl": 1.7294921875,
"learning_rate": 1e-06,
"loss": 0.0011,
"num_tokens": 11445405.0,
"reward": -2.562255859375,
"reward_std": 4.00094747543335,
"rewards/rm_reward_func/mean": -2.562255859375,
"rewards/rm_reward_func/std": 5.972076416015625,
"step": 867
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 499.0,
"completions/mean_length": 219.96875,
"completions/mean_terminated_length": 189.7586212158203,
"completions/min_length": 48.0,
"completions/min_terminated_length": 48.0,
"epoch": 0.6944,
"grad_norm": 21.208097457885742,
"kl": 4.55859375,
"learning_rate": 1e-06,
"loss": 0.1824,
"num_tokens": 11455276.0,
"reward": 0.225830078125,
"reward_std": 9.15473461151123,
"rewards/rm_reward_func/mean": 0.225830078125,
"rewards/rm_reward_func/std": 12.977889060974121,
"step": 868
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 479.0,
"completions/mean_length": 295.71875,
"completions/mean_terminated_length": 255.6666717529297,
"completions/min_length": 38.0,
"completions/min_terminated_length": 38.0,
"epoch": 0.6952,
"grad_norm": 61.403076171875,
"kl": 2.2958984375,
"learning_rate": 1e-06,
"loss": 0.0541,
"num_tokens": 11466739.0,
"reward": -4.762786865234375,
"reward_std": 12.194247245788574,
"rewards/rm_reward_func/mean": -4.762786865234375,
"rewards/rm_reward_func/std": 13.474750518798828,
"step": 869
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 387.0,
"completions/max_terminated_length": 387.0,
"completions/mean_length": 194.40625,
"completions/mean_terminated_length": 194.40625,
"completions/min_length": 44.0,
"completions/min_terminated_length": 44.0,
"epoch": 0.696,
"grad_norm": 11.072949409484863,
"kl": 2.7587890625,
"learning_rate": 1e-06,
"loss": 0.1886,
"num_tokens": 11477680.0,
"reward": -2.43963623046875,
"reward_std": 2.318965435028076,
"rewards/rm_reward_func/mean": -2.43963623046875,
"rewards/rm_reward_func/std": 8.259891510009766,
"step": 870
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 359.0,
"completions/max_terminated_length": 359.0,
"completions/mean_length": 120.15625,
"completions/mean_terminated_length": 120.15625,
"completions/min_length": 47.0,
"completions/min_terminated_length": 47.0,
"epoch": 0.6968,
"grad_norm": 22.81490135192871,
"kl": 2.748046875,
"learning_rate": 1e-06,
"loss": 0.1335,
"num_tokens": 11484381.0,
"reward": -4.06182861328125,
"reward_std": 3.879387855529785,
"rewards/rm_reward_func/mean": -4.06182861328125,
"rewards/rm_reward_func/std": 8.522046089172363,
"step": 871
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 470.0,
"completions/max_terminated_length": 470.0,
"completions/mean_length": 184.25,
"completions/mean_terminated_length": 184.25,
"completions/min_length": 54.0,
"completions/min_terminated_length": 54.0,
"epoch": 0.6976,
"grad_norm": 21.19168472290039,
"kl": 1.7607421875,
"learning_rate": 1e-06,
"loss": 0.0231,
"num_tokens": 11493045.0,
"reward": -6.307861328125,
"reward_std": 5.177760601043701,
"rewards/rm_reward_func/mean": -6.307861328125,
"rewards/rm_reward_func/std": 17.660837173461914,
"step": 872
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 491.0,
"completions/max_terminated_length": 491.0,
"completions/mean_length": 302.21875,
"completions/mean_terminated_length": 302.21875,
"completions/min_length": 65.0,
"completions/min_terminated_length": 65.0,
"epoch": 0.6984,
"grad_norm": 13.353934288024902,
"kl": 0.684326171875,
"learning_rate": 1e-06,
"loss": -0.0657,
"num_tokens": 11505260.0,
"reward": 6.2265625,
"reward_std": 4.352338790893555,
"rewards/rm_reward_func/mean": 6.2265625,
"rewards/rm_reward_func/std": 12.638473510742188,
"step": 873
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 469.0,
"completions/max_terminated_length": 469.0,
"completions/mean_length": 275.75,
"completions/mean_terminated_length": 275.75,
"completions/min_length": 78.0,
"completions/min_terminated_length": 78.0,
"epoch": 0.6992,
"grad_norm": 10.98495864868164,
"kl": 2.4599609375,
"learning_rate": 1e-06,
"loss": 0.0102,
"num_tokens": 11518476.0,
"reward": -2.28125,
"reward_std": 8.162097930908203,
"rewards/rm_reward_func/mean": -2.28125,
"rewards/rm_reward_func/std": 16.660022735595703,
"step": 874
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 507.0,
"completions/mean_length": 360.09375,
"completions/mean_terminated_length": 309.4583435058594,
"completions/min_length": 140.0,
"completions/min_terminated_length": 140.0,
"epoch": 0.7,
"grad_norm": 10.327630043029785,
"kl": 0.825439453125,
"learning_rate": 1e-06,
"loss": 0.0071,
"num_tokens": 11532407.0,
"reward": 5.72357177734375,
"reward_std": 6.062471389770508,
"rewards/rm_reward_func/mean": 5.72357177734375,
"rewards/rm_reward_func/std": 8.450596809387207,
"step": 875
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 505.0,
"completions/mean_length": 261.65625,
"completions/mean_terminated_length": 253.5806427001953,
"completions/min_length": 78.0,
"completions/min_terminated_length": 78.0,
"epoch": 0.7008,
"grad_norm": 12.771651268005371,
"kl": 0.5576171875,
"learning_rate": 1e-06,
"loss": -0.3164,
"num_tokens": 11544620.0,
"reward": -0.7434616088867188,
"reward_std": 7.011323928833008,
"rewards/rm_reward_func/mean": -0.7434616088867188,
"rewards/rm_reward_func/std": 9.20909595489502,
"step": 876
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 449.0,
"completions/mean_length": 357.53125,
"completions/mean_terminated_length": 306.04168701171875,
"completions/min_length": 45.0,
"completions/min_terminated_length": 45.0,
"epoch": 0.7016,
"grad_norm": 6.507169246673584,
"kl": 0.88671875,
"learning_rate": 1e-06,
"loss": 0.0546,
"num_tokens": 11558797.0,
"reward": 2.79498291015625,
"reward_std": 4.3798723220825195,
"rewards/rm_reward_func/mean": 2.79498291015625,
"rewards/rm_reward_func/std": 11.152603149414062,
"step": 877
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 459.0,
"completions/mean_length": 337.21875,
"completions/mean_terminated_length": 325.5666809082031,
"completions/min_length": 84.0,
"completions/min_terminated_length": 84.0,
"epoch": 0.7024,
"grad_norm": 7.747379302978516,
"kl": 2.21875,
"learning_rate": 1e-06,
"loss": -0.0081,
"num_tokens": 11572412.0,
"reward": -0.7412109375,
"reward_std": 10.57952880859375,
"rewards/rm_reward_func/mean": -0.7412109375,
"rewards/rm_reward_func/std": 18.887123107910156,
"step": 878
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 467.0,
"completions/mean_length": 245.84375,
"completions/mean_terminated_length": 218.3103485107422,
"completions/min_length": 71.0,
"completions/min_terminated_length": 71.0,
"epoch": 0.7032,
"grad_norm": 12.370904922485352,
"kl": 1.8740234375,
"learning_rate": 1e-06,
"loss": 0.0823,
"num_tokens": 11583719.0,
"reward": -5.319202423095703,
"reward_std": 8.250015258789062,
"rewards/rm_reward_func/mean": -5.319202423095703,
"rewards/rm_reward_func/std": 13.415038108825684,
"step": 879
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 501.0,
"completions/mean_length": 256.78125,
"completions/mean_terminated_length": 239.7666778564453,
"completions/min_length": 79.0,
"completions/min_terminated_length": 79.0,
"epoch": 0.704,
"grad_norm": 6.51011848449707,
"kl": 0.8095703125,
"learning_rate": 1e-06,
"loss": 0.0124,
"num_tokens": 11594712.0,
"reward": 12.4833984375,
"reward_std": 4.730257034301758,
"rewards/rm_reward_func/mean": 12.4833984375,
"rewards/rm_reward_func/std": 8.233878135681152,
"step": 880
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 502.0,
"completions/mean_length": 372.90625,
"completions/mean_terminated_length": 368.4193420410156,
"completions/min_length": 81.0,
"completions/min_terminated_length": 81.0,
"epoch": 0.7048,
"grad_norm": 6.078239440917969,
"kl": 0.372802734375,
"learning_rate": 1e-06,
"loss": -0.0806,
"num_tokens": 11608573.0,
"reward": 10.44952392578125,
"reward_std": 5.383957386016846,
"rewards/rm_reward_func/mean": 10.44952392578125,
"rewards/rm_reward_func/std": 14.256927490234375,
"step": 881
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 511.0,
"completions/mean_length": 276.0625,
"completions/mean_terminated_length": 251.65516662597656,
"completions/min_length": 48.0,
"completions/min_terminated_length": 48.0,
"epoch": 0.7056,
"grad_norm": 13.18306827545166,
"kl": 3.087646484375,
"learning_rate": 1e-06,
"loss": 0.2504,
"num_tokens": 11622719.0,
"reward": 2.23345947265625,
"reward_std": 10.839166641235352,
"rewards/rm_reward_func/mean": 2.23345947265625,
"rewards/rm_reward_func/std": 13.140902519226074,
"step": 882
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 79.0,
"completions/mean_length": 174.28125,
"completions/mean_terminated_length": 61.708335876464844,
"completions/min_length": 43.0,
"completions/min_terminated_length": 43.0,
"epoch": 0.7064,
"grad_norm": 9.840855598449707,
"kl": 0.5,
"learning_rate": 1e-06,
"loss": -0.0233,
"num_tokens": 11631200.0,
"reward": 2.555927276611328,
"reward_std": 1.7101624011993408,
"rewards/rm_reward_func/mean": 2.555927276611328,
"rewards/rm_reward_func/std": 6.12503719329834,
"step": 883
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 507.0,
"completions/mean_length": 397.53125,
"completions/mean_terminated_length": 385.6896667480469,
"completions/min_length": 110.0,
"completions/min_terminated_length": 110.0,
"epoch": 0.7072,
"grad_norm": 6.078858852386475,
"kl": 0.498046875,
"learning_rate": 1e-06,
"loss": -0.0905,
"num_tokens": 11647873.0,
"reward": 2.686767578125,
"reward_std": 7.326623439788818,
"rewards/rm_reward_func/mean": 2.686767578125,
"rewards/rm_reward_func/std": 11.423783302307129,
"step": 884
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 481.0,
"completions/mean_length": 377.34375,
"completions/mean_terminated_length": 339.6399841308594,
"completions/min_length": 124.0,
"completions/min_terminated_length": 124.0,
"epoch": 0.708,
"grad_norm": 10.225689888000488,
"kl": 2.7353515625,
"learning_rate": 1e-06,
"loss": 0.0753,
"num_tokens": 11662996.0,
"reward": 2.117431640625,
"reward_std": 9.586727142333984,
"rewards/rm_reward_func/mean": 2.117431640625,
"rewards/rm_reward_func/std": 12.49357795715332,
"step": 885
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 482.0,
"completions/mean_length": 235.8125,
"completions/mean_terminated_length": 196.35714721679688,
"completions/min_length": 30.0,
"completions/min_terminated_length": 30.0,
"epoch": 0.7088,
"grad_norm": 17.94410514831543,
"kl": 3.6484375,
"learning_rate": 1e-06,
"loss": 0.1815,
"num_tokens": 11672782.0,
"reward": -1.3177490234375,
"reward_std": 9.219121932983398,
"rewards/rm_reward_func/mean": -1.3177490234375,
"rewards/rm_reward_func/std": 11.736974716186523,
"step": 886
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 326.125,
"completions/mean_terminated_length": 264.16668701171875,
"completions/min_length": 52.0,
"completions/min_terminated_length": 52.0,
"epoch": 0.7096,
"grad_norm": 11.345696449279785,
"kl": 5.15625,
"learning_rate": 1e-06,
"loss": 0.192,
"num_tokens": 11690626.0,
"reward": 4.96630859375,
"reward_std": 6.32305908203125,
"rewards/rm_reward_func/mean": 4.96630859375,
"rewards/rm_reward_func/std": 13.138326644897461,
"step": 887
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 408.0,
"completions/max_terminated_length": 408.0,
"completions/mean_length": 187.46875,
"completions/mean_terminated_length": 187.46875,
"completions/min_length": 52.0,
"completions/min_terminated_length": 52.0,
"epoch": 0.7104,
"grad_norm": 21.07513427734375,
"kl": 0.44580078125,
"learning_rate": 1e-06,
"loss": -0.0441,
"num_tokens": 11699697.0,
"reward": 3.6982421875,
"reward_std": 3.48081636428833,
"rewards/rm_reward_func/mean": 3.6982421875,
"rewards/rm_reward_func/std": 14.544777870178223,
"step": 888
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 460.0,
"completions/mean_length": 328.15625,
"completions/mean_terminated_length": 322.2257995605469,
"completions/min_length": 56.0,
"completions/min_terminated_length": 56.0,
"epoch": 0.7112,
"grad_norm": 7.2959303855896,
"kl": 3.615478515625,
"learning_rate": 1e-06,
"loss": 0.152,
"num_tokens": 11712518.0,
"reward": 3.5261077880859375,
"reward_std": 9.151225090026855,
"rewards/rm_reward_func/mean": 3.5261077880859375,
"rewards/rm_reward_func/std": 11.036227226257324,
"step": 889
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 467.0,
"completions/mean_length": 306.21875,
"completions/mean_terminated_length": 268.1111145019531,
"completions/min_length": 81.0,
"completions/min_terminated_length": 81.0,
"epoch": 0.712,
"grad_norm": 7.229876518249512,
"kl": 4.206298828125,
"learning_rate": 1e-06,
"loss": 0.1888,
"num_tokens": 11726501.0,
"reward": 5.666015625,
"reward_std": 12.855541229248047,
"rewards/rm_reward_func/mean": 5.666015625,
"rewards/rm_reward_func/std": 19.8636531829834,
"step": 890
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 465.0,
"completions/mean_length": 177.25,
"completions/mean_terminated_length": 166.4516143798828,
"completions/min_length": 81.0,
"completions/min_terminated_length": 81.0,
"epoch": 0.7128,
"grad_norm": 11.873459815979004,
"kl": 4.671875,
"learning_rate": 1e-06,
"loss": 0.1815,
"num_tokens": 11735293.0,
"reward": -2.13092041015625,
"reward_std": 2.0769262313842773,
"rewards/rm_reward_func/mean": -2.13092041015625,
"rewards/rm_reward_func/std": 10.466885566711426,
"step": 891
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.40625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 511.0,
"completions/mean_length": 430.71875,
"completions/mean_terminated_length": 375.1052551269531,
"completions/min_length": 193.0,
"completions/min_terminated_length": 193.0,
"epoch": 0.7136,
"grad_norm": 15.590665817260742,
"kl": 9.9453125,
"learning_rate": 1e-06,
"loss": 0.3228,
"num_tokens": 11751772.0,
"reward": -10.29931640625,
"reward_std": 10.349671363830566,
"rewards/rm_reward_func/mean": -10.29931640625,
"rewards/rm_reward_func/std": 14.615714073181152,
"step": 892
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 477.0,
"completions/mean_length": 215.03125,
"completions/mean_terminated_length": 172.60714721679688,
"completions/min_length": 57.0,
"completions/min_terminated_length": 57.0,
"epoch": 0.7144,
"grad_norm": 23.91371726989746,
"kl": 7.96875,
"learning_rate": 1e-06,
"loss": 0.3347,
"num_tokens": 11762333.0,
"reward": -2.80712890625,
"reward_std": 3.1941628456115723,
"rewards/rm_reward_func/mean": -2.80712890625,
"rewards/rm_reward_func/std": 15.444784164428711,
"step": 893
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.40625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 505.0,
"completions/mean_length": 337.59375,
"completions/mean_terminated_length": 218.26315307617188,
"completions/min_length": 47.0,
"completions/min_terminated_length": 47.0,
"epoch": 0.7152,
"grad_norm": 15.916248321533203,
"kl": 10.64453125,
"learning_rate": 1e-06,
"loss": 0.4986,
"num_tokens": 11777384.0,
"reward": -6.3330078125,
"reward_std": 8.051085472106934,
"rewards/rm_reward_func/mean": -6.3330078125,
"rewards/rm_reward_func/std": 13.273330688476562,
"step": 894
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 504.0,
"completions/mean_length": 423.625,
"completions/mean_terminated_length": 389.0434875488281,
"completions/min_length": 141.0,
"completions/min_terminated_length": 141.0,
"epoch": 0.716,
"grad_norm": 27.08871078491211,
"kl": 11.65234375,
"learning_rate": 1e-06,
"loss": 0.4485,
"num_tokens": 11792972.0,
"reward": -5.36322021484375,
"reward_std": 8.532169342041016,
"rewards/rm_reward_func/mean": -5.36322021484375,
"rewards/rm_reward_func/std": 14.15001106262207,
"step": 895
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 454.0,
"completions/mean_length": 302.5,
"completions/mean_terminated_length": 232.6666717529297,
"completions/min_length": 87.0,
"completions/min_terminated_length": 87.0,
"epoch": 0.7168,
"grad_norm": 15.366225242614746,
"kl": 8.49609375,
"learning_rate": 1e-06,
"loss": 0.2878,
"num_tokens": 11804820.0,
"reward": -3.9163665771484375,
"reward_std": 10.814449310302734,
"rewards/rm_reward_func/mean": -3.9163665771484375,
"rewards/rm_reward_func/std": 13.626547813415527,
"step": 896
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 459.0,
"completions/mean_length": 301.5625,
"completions/mean_terminated_length": 262.59259033203125,
"completions/min_length": 87.0,
"completions/min_terminated_length": 87.0,
"epoch": 0.7176,
"grad_norm": 16.501802444458008,
"kl": 6.5625,
"learning_rate": 1e-06,
"loss": 0.3279,
"num_tokens": 11817710.0,
"reward": -13.09716796875,
"reward_std": 3.4331793785095215,
"rewards/rm_reward_func/mean": -13.09716796875,
"rewards/rm_reward_func/std": 7.296655178070068,
"step": 897
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 494.0,
"completions/mean_length": 383.15625,
"completions/mean_terminated_length": 305.8500061035156,
"completions/min_length": 81.0,
"completions/min_terminated_length": 81.0,
"epoch": 0.7184,
"grad_norm": 53.010684967041016,
"kl": 7.34423828125,
"learning_rate": 1e-06,
"loss": 0.2151,
"num_tokens": 11835283.0,
"reward": -2.9622802734375,
"reward_std": 7.178778648376465,
"rewards/rm_reward_func/mean": -2.9622802734375,
"rewards/rm_reward_func/std": 15.090096473693848,
"step": 898
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 429.0,
"completions/mean_length": 188.46875,
"completions/mean_terminated_length": 178.03225708007812,
"completions/min_length": 79.0,
"completions/min_terminated_length": 79.0,
"epoch": 0.7192,
"grad_norm": 12.646398544311523,
"kl": 2.4697265625,
"learning_rate": 1e-06,
"loss": 0.0837,
"num_tokens": 11844258.0,
"reward": -5.5667266845703125,
"reward_std": 3.9434971809387207,
"rewards/rm_reward_func/mean": -5.5667266845703125,
"rewards/rm_reward_func/std": 9.585037231445312,
"step": 899
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 479.0,
"completions/mean_length": 374.71875,
"completions/mean_terminated_length": 336.2799987792969,
"completions/min_length": 87.0,
"completions/min_terminated_length": 87.0,
"epoch": 0.72,
"grad_norm": 14.73233413696289,
"kl": 4.44140625,
"learning_rate": 1e-06,
"loss": 0.2017,
"num_tokens": 11858337.0,
"reward": 4.2802734375,
"reward_std": 4.078566074371338,
"rewards/rm_reward_func/mean": 4.2802734375,
"rewards/rm_reward_func/std": 18.690641403198242,
"step": 900
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 297.0625,
"completions/mean_terminated_length": 236.87998962402344,
"completions/min_length": 78.0,
"completions/min_terminated_length": 78.0,
"epoch": 0.7208,
"grad_norm": 12.702629089355469,
"kl": 1.1796875,
"learning_rate": 1e-06,
"loss": 0.0394,
"num_tokens": 11871475.0,
"reward": -1.2879791259765625,
"reward_std": 9.910463333129883,
"rewards/rm_reward_func/mean": -1.2879791259765625,
"rewards/rm_reward_func/std": 13.349971771240234,
"step": 901
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 477.0,
"completions/mean_length": 292.21875,
"completions/mean_terminated_length": 260.8214416503906,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.7216,
"grad_norm": 4.130387306213379,
"kl": 1.59130859375,
"learning_rate": 1e-06,
"loss": 0.0345,
"num_tokens": 11884650.0,
"reward": -2.958984375,
"reward_std": 4.643342971801758,
"rewards/rm_reward_func/mean": -2.958984375,
"rewards/rm_reward_func/std": 10.972875595092773,
"step": 902
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 469.0,
"completions/mean_length": 226.59375,
"completions/mean_terminated_length": 146.67999267578125,
"completions/min_length": 55.0,
"completions/min_terminated_length": 55.0,
"epoch": 0.7224,
"grad_norm": 7.769096851348877,
"kl": 3.14453125,
"learning_rate": 1e-06,
"loss": 0.1549,
"num_tokens": 11894645.0,
"reward": -5.154296875,
"reward_std": 1.5148166418075562,
"rewards/rm_reward_func/mean": -5.154296875,
"rewards/rm_reward_func/std": 8.897757530212402,
"step": 903
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 504.0,
"completions/mean_length": 368.5625,
"completions/mean_terminated_length": 335.4615478515625,
"completions/min_length": 99.0,
"completions/min_terminated_length": 99.0,
"epoch": 0.7232,
"grad_norm": 8.09273910522461,
"kl": 2.3486328125,
"learning_rate": 1e-06,
"loss": 0.146,
"num_tokens": 11908447.0,
"reward": -3.269775390625,
"reward_std": 5.145322799682617,
"rewards/rm_reward_func/mean": -3.269775390625,
"rewards/rm_reward_func/std": 9.46324634552002,
"step": 904
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 311.125,
"completions/mean_terminated_length": 264.76922607421875,
"completions/min_length": 81.0,
"completions/min_terminated_length": 81.0,
"epoch": 0.724,
"grad_norm": 8.956731796264648,
"kl": 2.220703125,
"learning_rate": 1e-06,
"loss": 0.0198,
"num_tokens": 11921539.0,
"reward": -2.416015625,
"reward_std": 7.836517333984375,
"rewards/rm_reward_func/mean": -2.416015625,
"rewards/rm_reward_func/std": 19.40036392211914,
"step": 905
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 491.0,
"completions/mean_length": 403.78125,
"completions/mean_terminated_length": 354.5909118652344,
"completions/min_length": 117.0,
"completions/min_terminated_length": 117.0,
"epoch": 0.7248,
"grad_norm": 5.295882701873779,
"kl": 3.423828125,
"learning_rate": 1e-06,
"loss": 0.1755,
"num_tokens": 11936668.0,
"reward": -7.770965576171875,
"reward_std": 7.4591522216796875,
"rewards/rm_reward_func/mean": -7.770965576171875,
"rewards/rm_reward_func/std": 9.990768432617188,
"step": 906
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.4375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 367.15625,
"completions/mean_terminated_length": 254.5,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.7256,
"grad_norm": 9.633946418762207,
"kl": 2.376953125,
"learning_rate": 1e-06,
"loss": 0.0554,
"num_tokens": 11952129.0,
"reward": -9.96728515625,
"reward_std": 5.492212772369385,
"rewards/rm_reward_func/mean": -9.96728515625,
"rewards/rm_reward_func/std": 7.280327320098877,
"step": 907
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 474.0,
"completions/mean_length": 175.5,
"completions/mean_terminated_length": 127.42857360839844,
"completions/min_length": 78.0,
"completions/min_terminated_length": 78.0,
"epoch": 0.7264,
"grad_norm": 4.560550212860107,
"kl": 0.552734375,
"learning_rate": 1e-06,
"loss": 0.0122,
"num_tokens": 11962097.0,
"reward": 7.42626953125,
"reward_std": 3.087636709213257,
"rewards/rm_reward_func/mean": 7.42626953125,
"rewards/rm_reward_func/std": 7.106700897216797,
"step": 908
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 497.0,
"completions/mean_length": 299.21875,
"completions/mean_terminated_length": 285.0333557128906,
"completions/min_length": 79.0,
"completions/min_terminated_length": 79.0,
"epoch": 0.7272,
"grad_norm": 7.450979232788086,
"kl": 1.3125,
"learning_rate": 1e-06,
"loss": 0.0609,
"num_tokens": 11974520.0,
"reward": 3.33935546875,
"reward_std": 8.265787124633789,
"rewards/rm_reward_func/mean": 3.33935546875,
"rewards/rm_reward_func/std": 13.333514213562012,
"step": 909
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 366.6875,
"completions/mean_terminated_length": 309.8260803222656,
"completions/min_length": 104.0,
"completions/min_terminated_length": 104.0,
"epoch": 0.728,
"grad_norm": 8.965705871582031,
"kl": 1.3427734375,
"learning_rate": 1e-06,
"loss": 0.0475,
"num_tokens": 11989574.0,
"reward": -3.5350265502929688,
"reward_std": 7.181827068328857,
"rewards/rm_reward_func/mean": -3.5350265502929688,
"rewards/rm_reward_func/std": 8.107301712036133,
"step": 910
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 476.0,
"completions/mean_length": 325.25,
"completions/mean_terminated_length": 290.6666564941406,
"completions/min_length": 124.0,
"completions/min_terminated_length": 124.0,
"epoch": 0.7288,
"grad_norm": 6.053767681121826,
"kl": 0.9969482421875,
"learning_rate": 1e-06,
"loss": 0.0931,
"num_tokens": 12003822.0,
"reward": -4.5394287109375,
"reward_std": 6.439982891082764,
"rewards/rm_reward_func/mean": -4.5394287109375,
"rewards/rm_reward_func/std": 12.76736831665039,
"step": 911
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.46875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 511.0,
"completions/mean_length": 380.8125,
"completions/mean_terminated_length": 265.058837890625,
"completions/min_length": 79.0,
"completions/min_terminated_length": 79.0,
"epoch": 0.7296,
"grad_norm": 13.645872116088867,
"kl": 1.7490234375,
"learning_rate": 1e-06,
"loss": 0.0605,
"num_tokens": 12019832.0,
"reward": -1.08984375,
"reward_std": 7.622467994689941,
"rewards/rm_reward_func/mean": -1.08984375,
"rewards/rm_reward_func/std": 14.811065673828125,
"step": 912
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 507.0,
"completions/mean_length": 312.78125,
"completions/mean_terminated_length": 266.8077087402344,
"completions/min_length": 78.0,
"completions/min_terminated_length": 78.0,
"epoch": 0.7304,
"grad_norm": 9.30956745147705,
"kl": 1.2353515625,
"learning_rate": 1e-06,
"loss": -0.0088,
"num_tokens": 12032193.0,
"reward": 1.819305419921875,
"reward_std": 11.730962753295898,
"rewards/rm_reward_func/mean": 1.819305419921875,
"rewards/rm_reward_func/std": 17.3555908203125,
"step": 913
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 403.78125,
"completions/mean_terminated_length": 367.7083435058594,
"completions/min_length": 86.0,
"completions/min_terminated_length": 86.0,
"epoch": 0.7312,
"grad_norm": 8.707047462463379,
"kl": 1.447509765625,
"learning_rate": 1e-06,
"loss": -0.0823,
"num_tokens": 12047786.0,
"reward": 1.6616268157958984,
"reward_std": 7.0725531578063965,
"rewards/rm_reward_func/mean": 1.6616268157958984,
"rewards/rm_reward_func/std": 11.841200828552246,
"step": 914
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 483.0,
"completions/mean_length": 325.09375,
"completions/mean_terminated_length": 305.75860595703125,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.732,
"grad_norm": 6.918102741241455,
"kl": 1.4365234375,
"learning_rate": 1e-06,
"loss": 0.0435,
"num_tokens": 12061229.0,
"reward": -0.952178955078125,
"reward_std": 10.511100769042969,
"rewards/rm_reward_func/mean": -0.952178955078125,
"rewards/rm_reward_func/std": 13.577516555786133,
"step": 915
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 502.0,
"completions/mean_length": 302.4375,
"completions/mean_terminated_length": 272.5,
"completions/min_length": 65.0,
"completions/min_terminated_length": 65.0,
"epoch": 0.7328,
"grad_norm": 16.212295532226562,
"kl": 1.50927734375,
"learning_rate": 1e-06,
"loss": 0.2198,
"num_tokens": 12073931.0,
"reward": -2.2662353515625,
"reward_std": 6.384040832519531,
"rewards/rm_reward_func/mean": -2.2662353515625,
"rewards/rm_reward_func/std": 8.968281745910645,
"step": 916
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 220.5625,
"completions/mean_terminated_length": 201.1333465576172,
"completions/min_length": 79.0,
"completions/min_terminated_length": 79.0,
"epoch": 0.7336,
"grad_norm": 6.973321914672852,
"kl": 0.521484375,
"learning_rate": 1e-06,
"loss": -0.0016,
"num_tokens": 12084549.0,
"reward": 6.8271484375,
"reward_std": 3.8583192825317383,
"rewards/rm_reward_func/mean": 6.8271484375,
"rewards/rm_reward_func/std": 11.556774139404297,
"step": 917
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 432.0,
"completions/max_terminated_length": 432.0,
"completions/mean_length": 193.75,
"completions/mean_terminated_length": 193.75,
"completions/min_length": 71.0,
"completions/min_terminated_length": 71.0,
"epoch": 0.7344,
"grad_norm": 11.093525886535645,
"kl": 0.890625,
"learning_rate": 1e-06,
"loss": -0.0353,
"num_tokens": 12093829.0,
"reward": 7.5962066650390625,
"reward_std": 5.227684497833252,
"rewards/rm_reward_func/mean": 7.5962066650390625,
"rewards/rm_reward_func/std": 7.640827178955078,
"step": 918
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 497.0,
"completions/mean_length": 263.03125,
"completions/mean_terminated_length": 246.433349609375,
"completions/min_length": 109.0,
"completions/min_terminated_length": 109.0,
"epoch": 0.7352,
"grad_norm": 8.780437469482422,
"kl": 1.28955078125,
"learning_rate": 1e-06,
"loss": -0.0192,
"num_tokens": 12105222.0,
"reward": 0.42479705810546875,
"reward_std": 5.950567722320557,
"rewards/rm_reward_func/mean": 0.42479705810546875,
"rewards/rm_reward_func/std": 10.092032432556152,
"step": 919
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 482.0,
"completions/mean_length": 341.21875,
"completions/mean_terminated_length": 284.29168701171875,
"completions/min_length": 64.0,
"completions/min_terminated_length": 64.0,
"epoch": 0.736,
"grad_norm": 18.315034866333008,
"kl": 2.19921875,
"learning_rate": 1e-06,
"loss": 0.3037,
"num_tokens": 12118829.0,
"reward": 7.9627838134765625,
"reward_std": 6.95778226852417,
"rewards/rm_reward_func/mean": 7.9627838134765625,
"rewards/rm_reward_func/std": 21.903451919555664,
"step": 920
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 507.0,
"completions/mean_length": 388.34375,
"completions/mean_terminated_length": 339.95654296875,
"completions/min_length": 56.0,
"completions/min_terminated_length": 56.0,
"epoch": 0.7368,
"grad_norm": 10.440993309020996,
"kl": 0.66943359375,
"learning_rate": 1e-06,
"loss": 0.1228,
"num_tokens": 12133104.0,
"reward": 8.855255126953125,
"reward_std": 10.921690940856934,
"rewards/rm_reward_func/mean": 8.855255126953125,
"rewards/rm_reward_func/std": 17.32367515563965,
"step": 921
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 476.0,
"completions/mean_length": 327.875,
"completions/mean_terminated_length": 285.3846130371094,
"completions/min_length": 93.0,
"completions/min_terminated_length": 93.0,
"epoch": 0.7376,
"grad_norm": 13.36120891571045,
"kl": 1.40380859375,
"learning_rate": 1e-06,
"loss": -0.0399,
"num_tokens": 12145796.0,
"reward": -1.751861572265625,
"reward_std": 5.661749839782715,
"rewards/rm_reward_func/mean": -1.751861572265625,
"rewards/rm_reward_func/std": 7.456704616546631,
"step": 922
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 511.0,
"completions/mean_length": 396.90625,
"completions/mean_terminated_length": 370.3461608886719,
"completions/min_length": 187.0,
"completions/min_terminated_length": 187.0,
"epoch": 0.7384,
"grad_norm": 7.57150936126709,
"kl": 4.10546875,
"learning_rate": 1e-06,
"loss": 0.0999,
"num_tokens": 12160545.0,
"reward": 2.98095703125,
"reward_std": 12.732685089111328,
"rewards/rm_reward_func/mean": 2.98095703125,
"rewards/rm_reward_func/std": 16.23927879333496,
"step": 923
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 496.0,
"completions/mean_length": 403.375,
"completions/mean_terminated_length": 360.86956787109375,
"completions/min_length": 141.0,
"completions/min_terminated_length": 141.0,
"epoch": 0.7392,
"grad_norm": 17.749086380004883,
"kl": 3.669921875,
"learning_rate": 1e-06,
"loss": 0.1162,
"num_tokens": 12175453.0,
"reward": -4.994476318359375,
"reward_std": 6.3849382400512695,
"rewards/rm_reward_func/mean": -4.994476318359375,
"rewards/rm_reward_func/std": 7.732586860656738,
"step": 924
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 375.03125,
"completions/mean_terminated_length": 349.6666564941406,
"completions/min_length": 159.0,
"completions/min_terminated_length": 159.0,
"epoch": 0.74,
"grad_norm": 10.675846099853516,
"kl": 2.8515625,
"learning_rate": 1e-06,
"loss": 0.0318,
"num_tokens": 12189734.0,
"reward": -0.42730712890625,
"reward_std": 8.244186401367188,
"rewards/rm_reward_func/mean": -0.42730712890625,
"rewards/rm_reward_func/std": 11.274063110351562,
"step": 925
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 501.0,
"completions/mean_length": 321.875,
"completions/mean_terminated_length": 258.5,
"completions/min_length": 68.0,
"completions/min_terminated_length": 68.0,
"epoch": 0.7408,
"grad_norm": 20.539968490600586,
"kl": 7.5546875,
"learning_rate": 1e-06,
"loss": 0.1656,
"num_tokens": 12202114.0,
"reward": -17.49755859375,
"reward_std": 5.544781684875488,
"rewards/rm_reward_func/mean": -17.49755859375,
"rewards/rm_reward_func/std": 8.136129379272461,
"step": 926
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 497.0,
"completions/mean_length": 215.375,
"completions/mean_terminated_length": 146.92308044433594,
"completions/min_length": 79.0,
"completions/min_terminated_length": 79.0,
"epoch": 0.7416,
"grad_norm": 16.78243637084961,
"kl": 1.171875,
"learning_rate": 1e-06,
"loss": 0.0322,
"num_tokens": 12214726.0,
"reward": -3.51953125,
"reward_std": 2.579955577850342,
"rewards/rm_reward_func/mean": -3.51953125,
"rewards/rm_reward_func/std": 12.52515697479248,
"step": 927
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 469.0,
"completions/mean_length": 340.34375,
"completions/mean_terminated_length": 292.2799987792969,
"completions/min_length": 87.0,
"completions/min_terminated_length": 87.0,
"epoch": 0.7424,
"grad_norm": 34.499900817871094,
"kl": 9.70751953125,
"learning_rate": 1e-06,
"loss": 0.3566,
"num_tokens": 12227649.0,
"reward": -9.71435546875,
"reward_std": 4.9702467918396,
"rewards/rm_reward_func/mean": -9.71435546875,
"rewards/rm_reward_func/std": 13.78485107421875,
"step": 928
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 340.21875,
"completions/mean_terminated_length": 308.40740966796875,
"completions/min_length": 67.0,
"completions/min_terminated_length": 67.0,
"epoch": 0.7432,
"grad_norm": 6.942431449890137,
"kl": 4.47021484375,
"learning_rate": 1e-06,
"loss": 0.0818,
"num_tokens": 12240512.0,
"reward": -4.043701171875,
"reward_std": 9.750587463378906,
"rewards/rm_reward_func/mean": -4.043701171875,
"rewards/rm_reward_func/std": 10.625539779663086,
"step": 929
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 423.0,
"completions/mean_length": 287.4375,
"completions/mean_terminated_length": 212.58334350585938,
"completions/min_length": 65.0,
"completions/min_terminated_length": 65.0,
"epoch": 0.744,
"grad_norm": 31.259197235107422,
"kl": 9.36083984375,
"learning_rate": 1e-06,
"loss": 0.3652,
"num_tokens": 12253814.0,
"reward": -10.76513671875,
"reward_std": 2.81522274017334,
"rewards/rm_reward_func/mean": -10.76513671875,
"rewards/rm_reward_func/std": 10.83386516571045,
"step": 930
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 469.0,
"completions/mean_length": 214.125,
"completions/mean_terminated_length": 194.2666778564453,
"completions/min_length": 64.0,
"completions/min_terminated_length": 64.0,
"epoch": 0.7448,
"grad_norm": 33.200462341308594,
"kl": 10.28125,
"learning_rate": 1e-06,
"loss": 0.465,
"num_tokens": 12265218.0,
"reward": -19.12255859375,
"reward_std": 3.9576473236083984,
"rewards/rm_reward_func/mean": -19.12255859375,
"rewards/rm_reward_func/std": 7.913651466369629,
"step": 931
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 496.0,
"completions/mean_length": 276.65625,
"completions/mean_terminated_length": 269.06451416015625,
"completions/min_length": 36.0,
"completions/min_terminated_length": 36.0,
"epoch": 0.7456,
"grad_norm": 22.295116424560547,
"kl": 6.87109375,
"learning_rate": 1e-06,
"loss": 0.1678,
"num_tokens": 12277919.0,
"reward": -6.76416015625,
"reward_std": 9.437423706054688,
"rewards/rm_reward_func/mean": -6.76416015625,
"rewards/rm_reward_func/std": 13.566336631774902,
"step": 932
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 490.0,
"completions/mean_length": 321.875,
"completions/mean_terminated_length": 294.71429443359375,
"completions/min_length": 116.0,
"completions/min_terminated_length": 116.0,
"epoch": 0.7464,
"grad_norm": 38.953163146972656,
"kl": 10.8125,
"learning_rate": 1e-06,
"loss": 0.3992,
"num_tokens": 12290051.0,
"reward": -13.24609375,
"reward_std": 5.0346527099609375,
"rewards/rm_reward_func/mean": -13.24609375,
"rewards/rm_reward_func/std": 5.71676778793335,
"step": 933
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 483.0,
"completions/mean_length": 240.125,
"completions/mean_terminated_length": 231.35482788085938,
"completions/min_length": 68.0,
"completions/min_terminated_length": 68.0,
"epoch": 0.7472,
"grad_norm": 16.947662353515625,
"kl": 5.8046875,
"learning_rate": 1e-06,
"loss": 0.1648,
"num_tokens": 12299375.0,
"reward": -9.80364990234375,
"reward_std": 7.922030448913574,
"rewards/rm_reward_func/mean": -9.80364990234375,
"rewards/rm_reward_func/std": 8.047533988952637,
"step": 934
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 493.0,
"completions/mean_length": 334.5,
"completions/mean_terminated_length": 265.0434875488281,
"completions/min_length": 81.0,
"completions/min_terminated_length": 81.0,
"epoch": 0.748,
"grad_norm": 20.085559844970703,
"kl": 3.55517578125,
"learning_rate": 1e-06,
"loss": 0.0963,
"num_tokens": 12313439.0,
"reward": -4.38671875,
"reward_std": 5.369791030883789,
"rewards/rm_reward_func/mean": -4.38671875,
"rewards/rm_reward_func/std": 12.364655494689941,
"step": 935
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 478.0,
"completions/max_terminated_length": 478.0,
"completions/mean_length": 190.5625,
"completions/mean_terminated_length": 190.5625,
"completions/min_length": 32.0,
"completions/min_terminated_length": 32.0,
"epoch": 0.7488,
"grad_norm": 19.231365203857422,
"kl": 5.12109375,
"learning_rate": 1e-06,
"loss": 0.3493,
"num_tokens": 12322225.0,
"reward": -9.181640625,
"reward_std": 6.0393853187561035,
"rewards/rm_reward_func/mean": -9.181640625,
"rewards/rm_reward_func/std": 11.844439506530762,
"step": 936
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 376.625,
"completions/mean_terminated_length": 331.5,
"completions/min_length": 41.0,
"completions/min_terminated_length": 41.0,
"epoch": 0.7496,
"grad_norm": 12.45418930053711,
"kl": 6.169921875,
"learning_rate": 1e-06,
"loss": 0.336,
"num_tokens": 12338125.0,
"reward": -4.42236328125,
"reward_std": 8.240564346313477,
"rewards/rm_reward_func/mean": -4.42236328125,
"rewards/rm_reward_func/std": 17.537168502807617,
"step": 937
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 455.0,
"completions/mean_length": 244.09375,
"completions/mean_terminated_length": 216.37930297851562,
"completions/min_length": 68.0,
"completions/min_terminated_length": 68.0,
"epoch": 0.7504,
"grad_norm": 15.027655601501465,
"kl": 2.1923828125,
"learning_rate": 1e-06,
"loss": 0.0436,
"num_tokens": 12351256.0,
"reward": -4.74468994140625,
"reward_std": 5.563222885131836,
"rewards/rm_reward_func/mean": -4.74468994140625,
"rewards/rm_reward_func/std": 10.182250022888184,
"step": 938
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 455.0,
"completions/mean_length": 273.71875,
"completions/mean_terminated_length": 249.0689697265625,
"completions/min_length": 81.0,
"completions/min_terminated_length": 81.0,
"epoch": 0.7512,
"grad_norm": 6.310357570648193,
"kl": 2.48291015625,
"learning_rate": 1e-06,
"loss": 0.1097,
"num_tokens": 12362543.0,
"reward": -3.814697265625,
"reward_std": 5.3505682945251465,
"rewards/rm_reward_func/mean": -3.814697265625,
"rewards/rm_reward_func/std": 7.993979454040527,
"step": 939
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 448.0,
"completions/mean_length": 170.59375,
"completions/mean_terminated_length": 147.83334350585938,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.752,
"grad_norm": 10.239607810974121,
"kl": 1.9765625,
"learning_rate": 1e-06,
"loss": 0.2201,
"num_tokens": 12370810.0,
"reward": 6.2581787109375,
"reward_std": 5.198629379272461,
"rewards/rm_reward_func/mean": 6.2581787109375,
"rewards/rm_reward_func/std": 11.37491512298584,
"step": 940
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 481.0,
"completions/mean_length": 358.1875,
"completions/mean_terminated_length": 342.2758483886719,
"completions/min_length": 85.0,
"completions/min_terminated_length": 85.0,
"epoch": 0.7528,
"grad_norm": 6.657660961151123,
"kl": 2.30615234375,
"learning_rate": 1e-06,
"loss": 0.0582,
"num_tokens": 12384584.0,
"reward": 1.087982177734375,
"reward_std": 10.667032241821289,
"rewards/rm_reward_func/mean": 1.087982177734375,
"rewards/rm_reward_func/std": 15.517912864685059,
"step": 941
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 487.0,
"completions/max_terminated_length": 487.0,
"completions/mean_length": 243.21875,
"completions/mean_terminated_length": 243.21875,
"completions/min_length": 81.0,
"completions/min_terminated_length": 81.0,
"epoch": 0.7536,
"grad_norm": 7.709587574005127,
"kl": 2.3125,
"learning_rate": 1e-06,
"loss": 0.123,
"num_tokens": 12396471.0,
"reward": -9.607421875,
"reward_std": 2.2891464233398438,
"rewards/rm_reward_func/mean": -9.607421875,
"rewards/rm_reward_func/std": 11.87063217163086,
"step": 942
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 487.0,
"completions/mean_length": 330.40625,
"completions/mean_terminated_length": 296.77777099609375,
"completions/min_length": 124.0,
"completions/min_terminated_length": 124.0,
"epoch": 0.7544,
"grad_norm": 9.007935523986816,
"kl": 0.546142578125,
"learning_rate": 1e-06,
"loss": 0.0883,
"num_tokens": 12412252.0,
"reward": 3.67486572265625,
"reward_std": 3.7311360836029053,
"rewards/rm_reward_func/mean": 3.67486572265625,
"rewards/rm_reward_func/std": 8.098952293395996,
"step": 943
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 447.0,
"completions/max_terminated_length": 447.0,
"completions/mean_length": 180.96875,
"completions/mean_terminated_length": 180.96875,
"completions/min_length": 41.0,
"completions/min_terminated_length": 41.0,
"epoch": 0.7552,
"grad_norm": 7.097393989562988,
"kl": 1.11328125,
"learning_rate": 1e-06,
"loss": 0.1134,
"num_tokens": 12421491.0,
"reward": 3.06976318359375,
"reward_std": 3.404435873031616,
"rewards/rm_reward_func/mean": 3.06976318359375,
"rewards/rm_reward_func/std": 5.2598958015441895,
"step": 944
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 474.0,
"completions/mean_length": 225.625,
"completions/mean_terminated_length": 206.53334045410156,
"completions/min_length": 81.0,
"completions/min_terminated_length": 81.0,
"epoch": 0.756,
"grad_norm": 6.910060405731201,
"kl": 0.96484375,
"learning_rate": 1e-06,
"loss": -0.0059,
"num_tokens": 12432391.0,
"reward": 1.87451171875,
"reward_std": 6.181256294250488,
"rewards/rm_reward_func/mean": 1.87451171875,
"rewards/rm_reward_func/std": 9.241499900817871,
"step": 945
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 264.9375,
"completions/mean_terminated_length": 168.26087951660156,
"completions/min_length": 41.0,
"completions/min_terminated_length": 41.0,
"epoch": 0.7568,
"grad_norm": 18.712726593017578,
"kl": 0.5518798828125,
"learning_rate": 1e-06,
"loss": 0.1766,
"num_tokens": 12446741.0,
"reward": -6.08203125,
"reward_std": 6.159107685089111,
"rewards/rm_reward_func/mean": -6.08203125,
"rewards/rm_reward_func/std": 11.603202819824219,
"step": 946
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 504.0,
"completions/mean_length": 298.78125,
"completions/mean_terminated_length": 276.7241516113281,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.7576,
"grad_norm": 6.926180362701416,
"kl": 0.4462890625,
"learning_rate": 1e-06,
"loss": 0.0145,
"num_tokens": 12458686.0,
"reward": 6.00732421875,
"reward_std": 3.4000492095947266,
"rewards/rm_reward_func/mean": 6.00732421875,
"rewards/rm_reward_func/std": 8.406200408935547,
"step": 947
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 415.0,
"completions/mean_length": 269.1875,
"completions/mean_terminated_length": 234.50001525878906,
"completions/min_length": 65.0,
"completions/min_terminated_length": 65.0,
"epoch": 0.7584,
"grad_norm": 7.784516334533691,
"kl": 2.767578125,
"learning_rate": 1e-06,
"loss": -0.0341,
"num_tokens": 12469732.0,
"reward": -5.19873046875,
"reward_std": 12.44931697845459,
"rewards/rm_reward_func/mean": -5.19873046875,
"rewards/rm_reward_func/std": 13.344382286071777,
"step": 948
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 468.0,
"completions/mean_length": 263.5,
"completions/mean_terminated_length": 246.933349609375,
"completions/min_length": 50.0,
"completions/min_terminated_length": 50.0,
"epoch": 0.7592,
"grad_norm": 7.466695785522461,
"kl": 1.8359375,
"learning_rate": 1e-06,
"loss": 0.0924,
"num_tokens": 12483748.0,
"reward": -2.9599609375,
"reward_std": 8.385852813720703,
"rewards/rm_reward_func/mean": -2.9599609375,
"rewards/rm_reward_func/std": 15.177970886230469,
"step": 949
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 507.0,
"completions/mean_length": 242.59375,
"completions/mean_terminated_length": 233.90321350097656,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.76,
"grad_norm": 14.014596939086914,
"kl": 1.61474609375,
"learning_rate": 1e-06,
"loss": -0.0958,
"num_tokens": 12494015.0,
"reward": 1.893310546875,
"reward_std": 7.712099075317383,
"rewards/rm_reward_func/mean": 1.893310546875,
"rewards/rm_reward_func/std": 14.392072677612305,
"step": 950
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 370.1875,
"completions/mean_terminated_length": 314.6956481933594,
"completions/min_length": 81.0,
"completions/min_terminated_length": 81.0,
"epoch": 0.7608,
"grad_norm": 9.24715805053711,
"kl": 1.931640625,
"learning_rate": 1e-06,
"loss": 0.0885,
"num_tokens": 12508949.0,
"reward": -0.7783203125,
"reward_std": 8.970268249511719,
"rewards/rm_reward_func/mean": -0.7783203125,
"rewards/rm_reward_func/std": 15.02639102935791,
"step": 951
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 474.0,
"completions/mean_length": 297.5625,
"completions/mean_terminated_length": 290.6451416015625,
"completions/min_length": 16.0,
"completions/min_terminated_length": 16.0,
"epoch": 0.7616,
"grad_norm": 13.282108306884766,
"kl": 2.541015625,
"learning_rate": 1e-06,
"loss": 0.0898,
"num_tokens": 12522735.0,
"reward": 3.421630859375,
"reward_std": 8.37659740447998,
"rewards/rm_reward_func/mean": 3.421630859375,
"rewards/rm_reward_func/std": 12.819450378417969,
"step": 952
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 493.0,
"completions/mean_length": 328.65625,
"completions/mean_terminated_length": 322.7419128417969,
"completions/min_length": 101.0,
"completions/min_terminated_length": 101.0,
"epoch": 0.7624,
"grad_norm": 9.377727508544922,
"kl": 1.41259765625,
"learning_rate": 1e-06,
"loss": -0.0102,
"num_tokens": 12535252.0,
"reward": 4.1376953125,
"reward_std": 6.576689720153809,
"rewards/rm_reward_func/mean": 4.1376953125,
"rewards/rm_reward_func/std": 16.938846588134766,
"step": 953
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 474.0,
"completions/mean_length": 379.1875,
"completions/mean_terminated_length": 360.21429443359375,
"completions/min_length": 68.0,
"completions/min_terminated_length": 68.0,
"epoch": 0.7632,
"grad_norm": 13.483288764953613,
"kl": 1.65625,
"learning_rate": 1e-06,
"loss": -0.0113,
"num_tokens": 12549658.0,
"reward": 4.1842041015625,
"reward_std": 11.937593460083008,
"rewards/rm_reward_func/mean": 4.1842041015625,
"rewards/rm_reward_func/std": 16.32895851135254,
"step": 954
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 499.0,
"completions/mean_length": 351.65625,
"completions/mean_terminated_length": 340.9666748046875,
"completions/min_length": 60.0,
"completions/min_terminated_length": 60.0,
"epoch": 0.764,
"grad_norm": 7.8393707275390625,
"kl": 2.38818359375,
"learning_rate": 1e-06,
"loss": 0.0489,
"num_tokens": 12563167.0,
"reward": -6.7680511474609375,
"reward_std": 6.109908103942871,
"rewards/rm_reward_func/mean": -6.7680511474609375,
"rewards/rm_reward_func/std": 9.325031280517578,
"step": 955
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 507.0,
"completions/max_terminated_length": 507.0,
"completions/mean_length": 253.34375,
"completions/mean_terminated_length": 253.34375,
"completions/min_length": 81.0,
"completions/min_terminated_length": 81.0,
"epoch": 0.7648,
"grad_norm": 6.62860631942749,
"kl": 0.927734375,
"learning_rate": 1e-06,
"loss": -0.0399,
"num_tokens": 12576426.0,
"reward": -0.5888671875,
"reward_std": 4.022240161895752,
"rewards/rm_reward_func/mean": -0.5888671875,
"rewards/rm_reward_func/std": 6.941095352172852,
"step": 956
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 491.0,
"completions/mean_length": 290.53125,
"completions/mean_terminated_length": 267.6206970214844,
"completions/min_length": 110.0,
"completions/min_terminated_length": 110.0,
"epoch": 0.7656,
"grad_norm": 9.683281898498535,
"kl": 4.0126953125,
"learning_rate": 1e-06,
"loss": 0.0822,
"num_tokens": 12587875.0,
"reward": -11.738689422607422,
"reward_std": 5.651496887207031,
"rewards/rm_reward_func/mean": -11.738689422607422,
"rewards/rm_reward_func/std": 10.464712142944336,
"step": 957
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 452.0,
"completions/mean_length": 278.40625,
"completions/mean_terminated_length": 254.2413787841797,
"completions/min_length": 47.0,
"completions/min_terminated_length": 47.0,
"epoch": 0.7664,
"grad_norm": 13.09759521484375,
"kl": 2.10693359375,
"learning_rate": 1e-06,
"loss": 0.0637,
"num_tokens": 12599120.0,
"reward": -3.7507591247558594,
"reward_std": 5.601696014404297,
"rewards/rm_reward_func/mean": -3.7507591247558594,
"rewards/rm_reward_func/std": 7.122979640960693,
"step": 958
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 502.0,
"completions/mean_length": 385.4375,
"completions/mean_terminated_length": 362.0,
"completions/min_length": 73.0,
"completions/min_terminated_length": 73.0,
"epoch": 0.7672,
"grad_norm": 17.111623764038086,
"kl": 2.60302734375,
"learning_rate": 1e-06,
"loss": 0.0545,
"num_tokens": 12614662.0,
"reward": 11.8017578125,
"reward_std": 12.20899772644043,
"rewards/rm_reward_func/mean": 11.8017578125,
"rewards/rm_reward_func/std": 16.13597297668457,
"step": 959
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 476.0,
"completions/mean_length": 285.59375,
"completions/mean_terminated_length": 270.5,
"completions/min_length": 67.0,
"completions/min_terminated_length": 67.0,
"epoch": 0.768,
"grad_norm": 18.32476806640625,
"kl": 2.861328125,
"learning_rate": 1e-06,
"loss": 0.1135,
"num_tokens": 12628177.0,
"reward": 10.161865234375,
"reward_std": 9.773149490356445,
"rewards/rm_reward_func/mean": 10.161865234375,
"rewards/rm_reward_func/std": 18.04802131652832,
"step": 960
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 511.0,
"completions/mean_length": 314.96875,
"completions/mean_terminated_length": 294.5862121582031,
"completions/min_length": 67.0,
"completions/min_terminated_length": 67.0,
"epoch": 0.7688,
"grad_norm": 11.534658432006836,
"kl": 2.671875,
"learning_rate": 1e-06,
"loss": 0.1081,
"num_tokens": 12640928.0,
"reward": -3.29248046875,
"reward_std": 11.981504440307617,
"rewards/rm_reward_func/mean": -3.29248046875,
"rewards/rm_reward_func/std": 14.41004753112793,
"step": 961
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 430.0,
"completions/max_terminated_length": 430.0,
"completions/mean_length": 194.5,
"completions/mean_terminated_length": 194.5,
"completions/min_length": 81.0,
"completions/min_terminated_length": 81.0,
"epoch": 0.7696,
"grad_norm": 6.043729782104492,
"kl": 0.4736328125,
"learning_rate": 1e-06,
"loss": -0.0156,
"num_tokens": 12653056.0,
"reward": -7.85498046875,
"reward_std": 1.4552382230758667,
"rewards/rm_reward_func/mean": -7.85498046875,
"rewards/rm_reward_func/std": 5.594080924987793,
"step": 962
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 284.84375,
"completions/mean_terminated_length": 269.70001220703125,
"completions/min_length": 55.0,
"completions/min_terminated_length": 55.0,
"epoch": 0.7704,
"grad_norm": 9.446344375610352,
"kl": 1.1533203125,
"learning_rate": 1e-06,
"loss": 0.0328,
"num_tokens": 12664139.0,
"reward": 1.72314453125,
"reward_std": 7.9739813804626465,
"rewards/rm_reward_func/mean": 1.72314453125,
"rewards/rm_reward_func/std": 14.244680404663086,
"step": 963
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 287.34375,
"completions/mean_terminated_length": 280.0967712402344,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.7712,
"grad_norm": 9.243260383605957,
"kl": 0.39892578125,
"learning_rate": 1e-06,
"loss": -0.0251,
"num_tokens": 12675110.0,
"reward": 13.383621215820312,
"reward_std": 4.858243942260742,
"rewards/rm_reward_func/mean": 13.383621215820312,
"rewards/rm_reward_func/std": 12.968853950500488,
"step": 964
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 404.0,
"completions/mean_length": 262.125,
"completions/mean_terminated_length": 192.1599884033203,
"completions/min_length": 76.0,
"completions/min_terminated_length": 76.0,
"epoch": 0.772,
"grad_norm": 13.435283660888672,
"kl": 4.7197265625,
"learning_rate": 1e-06,
"loss": 0.2661,
"num_tokens": 12689778.0,
"reward": -4.897491455078125,
"reward_std": 4.347750663757324,
"rewards/rm_reward_func/mean": -4.897491455078125,
"rewards/rm_reward_func/std": 11.386714935302734,
"step": 965
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 501.0,
"completions/mean_length": 386.4375,
"completions/mean_terminated_length": 344.5833435058594,
"completions/min_length": 201.0,
"completions/min_terminated_length": 201.0,
"epoch": 0.7728,
"grad_norm": 6.507021903991699,
"kl": 1.4296875,
"learning_rate": 1e-06,
"loss": -0.0111,
"num_tokens": 12705064.0,
"reward": 13.21240234375,
"reward_std": 10.414816856384277,
"rewards/rm_reward_func/mean": 13.21240234375,
"rewards/rm_reward_func/std": 14.156932830810547,
"step": 966
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 473.0,
"completions/max_terminated_length": 473.0,
"completions/mean_length": 306.90625,
"completions/mean_terminated_length": 306.90625,
"completions/min_length": 108.0,
"completions/min_terminated_length": 108.0,
"epoch": 0.7736,
"grad_norm": 6.9047393798828125,
"kl": 2.828125,
"learning_rate": 1e-06,
"loss": -0.0205,
"num_tokens": 12716701.0,
"reward": 2.09149169921875,
"reward_std": 10.416152954101562,
"rewards/rm_reward_func/mean": 2.09149169921875,
"rewards/rm_reward_func/std": 12.270896911621094,
"step": 967
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 289.03125,
"completions/mean_terminated_length": 281.8387145996094,
"completions/min_length": 76.0,
"completions/min_terminated_length": 76.0,
"epoch": 0.7744,
"grad_norm": 7.523399353027344,
"kl": 2.68798828125,
"learning_rate": 1e-06,
"loss": 0.0547,
"num_tokens": 12729430.0,
"reward": 2.884521484375,
"reward_std": 5.321774482727051,
"rewards/rm_reward_func/mean": 2.884521484375,
"rewards/rm_reward_func/std": 15.252958297729492,
"step": 968
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 423.0,
"completions/mean_length": 210.21875,
"completions/mean_terminated_length": 200.48387145996094,
"completions/min_length": 57.0,
"completions/min_terminated_length": 57.0,
"epoch": 0.7752,
"grad_norm": 30.35150909423828,
"kl": 11.3837890625,
"learning_rate": 1e-06,
"loss": 0.6159,
"num_tokens": 12740117.0,
"reward": -4.417724609375,
"reward_std": 6.602893352508545,
"rewards/rm_reward_func/mean": -4.417724609375,
"rewards/rm_reward_func/std": 15.255743980407715,
"step": 969
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 500.0,
"completions/mean_length": 353.03125,
"completions/mean_terminated_length": 336.5862121582031,
"completions/min_length": 107.0,
"completions/min_terminated_length": 107.0,
"epoch": 0.776,
"grad_norm": 8.360908508300781,
"kl": 2.9921875,
"learning_rate": 1e-06,
"loss": 0.1711,
"num_tokens": 12753542.0,
"reward": -0.7406158447265625,
"reward_std": 4.496933937072754,
"rewards/rm_reward_func/mean": -0.7406158447265625,
"rewards/rm_reward_func/std": 11.36292839050293,
"step": 970
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 360.15625,
"completions/mean_terminated_length": 291.1363830566406,
"completions/min_length": 51.0,
"completions/min_terminated_length": 51.0,
"epoch": 0.7768,
"grad_norm": 10.487771987915039,
"kl": 0.97509765625,
"learning_rate": 1e-06,
"loss": 0.0275,
"num_tokens": 12767451.0,
"reward": 5.02734375,
"reward_std": 7.548166275024414,
"rewards/rm_reward_func/mean": 5.02734375,
"rewards/rm_reward_func/std": 21.371307373046875,
"step": 971
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 505.0,
"completions/mean_length": 348.0625,
"completions/mean_terminated_length": 331.10345458984375,
"completions/min_length": 96.0,
"completions/min_terminated_length": 96.0,
"epoch": 0.7776,
"grad_norm": 6.754936695098877,
"kl": 2.9931640625,
"learning_rate": 1e-06,
"loss": 0.0756,
"num_tokens": 12780813.0,
"reward": 0.3992156982421875,
"reward_std": 5.77485466003418,
"rewards/rm_reward_func/mean": 0.3992156982421875,
"rewards/rm_reward_func/std": 6.508362770080566,
"step": 972
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 477.0,
"completions/mean_length": 350.78125,
"completions/mean_terminated_length": 345.58062744140625,
"completions/min_length": 72.0,
"completions/min_terminated_length": 72.0,
"epoch": 0.7784,
"grad_norm": 6.786907196044922,
"kl": 1.45751953125,
"learning_rate": 1e-06,
"loss": -0.0457,
"num_tokens": 12797526.0,
"reward": 14.609710693359375,
"reward_std": 11.176668167114258,
"rewards/rm_reward_func/mean": 14.609710693359375,
"rewards/rm_reward_func/std": 18.939729690551758,
"step": 973
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 499.0,
"completions/mean_length": 307.34375,
"completions/mean_terminated_length": 250.0399932861328,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.7792,
"grad_norm": 9.214887619018555,
"kl": 2.23046875,
"learning_rate": 1e-06,
"loss": 0.0533,
"num_tokens": 12811017.0,
"reward": -0.3729248046875,
"reward_std": 4.81978702545166,
"rewards/rm_reward_func/mean": -0.3729248046875,
"rewards/rm_reward_func/std": 6.530143737792969,
"step": 974
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 476.0,
"completions/mean_length": 307.28125,
"completions/mean_terminated_length": 293.63336181640625,
"completions/min_length": 112.0,
"completions/min_terminated_length": 112.0,
"epoch": 0.78,
"grad_norm": 28.01193618774414,
"kl": 11.6064453125,
"learning_rate": 1e-06,
"loss": 0.4855,
"num_tokens": 12823162.0,
"reward": -6.5583343505859375,
"reward_std": 6.375615119934082,
"rewards/rm_reward_func/mean": -6.5583343505859375,
"rewards/rm_reward_func/std": 13.018743515014648,
"step": 975
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 496.0,
"completions/mean_length": 359.5,
"completions/mean_terminated_length": 331.2592468261719,
"completions/min_length": 148.0,
"completions/min_terminated_length": 148.0,
"epoch": 0.7808,
"grad_norm": 16.22539710998535,
"kl": 6.41015625,
"learning_rate": 1e-06,
"loss": 0.1488,
"num_tokens": 12837546.0,
"reward": -6.0869140625,
"reward_std": 5.604630470275879,
"rewards/rm_reward_func/mean": -6.0869140625,
"rewards/rm_reward_func/std": 11.859865188598633,
"step": 976
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 485.0,
"completions/mean_length": 297.25,
"completions/mean_terminated_length": 237.1199951171875,
"completions/min_length": 63.0,
"completions/min_terminated_length": 63.0,
"epoch": 0.7816,
"grad_norm": 43.490577697753906,
"kl": 7.90283203125,
"learning_rate": 1e-06,
"loss": 0.1253,
"num_tokens": 12850946.0,
"reward": -4.4287109375,
"reward_std": 7.8558669090271,
"rewards/rm_reward_func/mean": -4.4287109375,
"rewards/rm_reward_func/std": 18.278430938720703,
"step": 977
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 333.0,
"completions/max_terminated_length": 333.0,
"completions/mean_length": 167.125,
"completions/mean_terminated_length": 167.125,
"completions/min_length": 33.0,
"completions/min_terminated_length": 33.0,
"epoch": 0.7824,
"grad_norm": 9.18833065032959,
"kl": 2.99072265625,
"learning_rate": 1e-06,
"loss": 0.2724,
"num_tokens": 12858686.0,
"reward": 2.391876220703125,
"reward_std": 3.7164864540100098,
"rewards/rm_reward_func/mean": 2.391876220703125,
"rewards/rm_reward_func/std": 9.332040786743164,
"step": 978
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 490.0,
"completions/mean_length": 340.0625,
"completions/mean_terminated_length": 334.51611328125,
"completions/min_length": 37.0,
"completions/min_terminated_length": 37.0,
"epoch": 0.7832,
"grad_norm": 6.65054988861084,
"kl": 0.5810546875,
"learning_rate": 1e-06,
"loss": -0.0442,
"num_tokens": 12872840.0,
"reward": 8.1917724609375,
"reward_std": 7.095458984375,
"rewards/rm_reward_func/mean": 8.1917724609375,
"rewards/rm_reward_func/std": 12.081109046936035,
"step": 979
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 505.0,
"completions/mean_length": 403.0,
"completions/mean_terminated_length": 382.8148193359375,
"completions/min_length": 76.0,
"completions/min_terminated_length": 76.0,
"epoch": 0.784,
"grad_norm": 9.431859970092773,
"kl": 7.0546875,
"learning_rate": 1e-06,
"loss": 0.2173,
"num_tokens": 12888208.0,
"reward": 2.475616455078125,
"reward_std": 8.312936782836914,
"rewards/rm_reward_func/mean": 2.475616455078125,
"rewards/rm_reward_func/std": 13.530970573425293,
"step": 980
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 407.84375,
"completions/mean_terminated_length": 383.8077087402344,
"completions/min_length": 82.0,
"completions/min_terminated_length": 82.0,
"epoch": 0.7848,
"grad_norm": 6.22176456451416,
"kl": 3.5087890625,
"learning_rate": 1e-06,
"loss": 0.0194,
"num_tokens": 12907155.0,
"reward": 4.936767578125,
"reward_std": 9.09925651550293,
"rewards/rm_reward_func/mean": 4.936767578125,
"rewards/rm_reward_func/std": 20.013263702392578,
"step": 981
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 496.0,
"completions/mean_length": 320.96875,
"completions/mean_terminated_length": 301.2069091796875,
"completions/min_length": 78.0,
"completions/min_terminated_length": 78.0,
"epoch": 0.7856,
"grad_norm": 5.159428119659424,
"kl": 1.8134765625,
"learning_rate": 1e-06,
"loss": 0.0296,
"num_tokens": 12921042.0,
"reward": 9.160308837890625,
"reward_std": 9.23448371887207,
"rewards/rm_reward_func/mean": 9.160308837890625,
"rewards/rm_reward_func/std": 13.724321365356445,
"step": 982
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 426.0,
"completions/max_terminated_length": 426.0,
"completions/mean_length": 257.75,
"completions/mean_terminated_length": 257.75,
"completions/min_length": 46.0,
"completions/min_terminated_length": 46.0,
"epoch": 0.7864,
"grad_norm": 10.37830924987793,
"kl": 1.292724609375,
"learning_rate": 1e-06,
"loss": -0.0107,
"num_tokens": 12932818.0,
"reward": 3.912353515625,
"reward_std": 6.3594069480896,
"rewards/rm_reward_func/mean": 3.912353515625,
"rewards/rm_reward_func/std": 14.007357597351074,
"step": 983
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.53125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 390.03125,
"completions/mean_terminated_length": 251.80001831054688,
"completions/min_length": 81.0,
"completions/min_terminated_length": 81.0,
"epoch": 0.7872,
"grad_norm": 8.068547248840332,
"kl": 0.436279296875,
"learning_rate": 1e-06,
"loss": 0.0255,
"num_tokens": 12951635.0,
"reward": 4.229248046875,
"reward_std": 4.562774658203125,
"rewards/rm_reward_func/mean": 4.229248046875,
"rewards/rm_reward_func/std": 10.442963600158691,
"step": 984
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 507.0,
"completions/mean_length": 262.125,
"completions/mean_terminated_length": 204.4615478515625,
"completions/min_length": 63.0,
"completions/min_terminated_length": 63.0,
"epoch": 0.788,
"grad_norm": 11.331951141357422,
"kl": 0.728759765625,
"learning_rate": 1e-06,
"loss": 0.0525,
"num_tokens": 12965559.0,
"reward": 3.13623046875,
"reward_std": 7.865469932556152,
"rewards/rm_reward_func/mean": 3.13623046875,
"rewards/rm_reward_func/std": 14.55949592590332,
"step": 985
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 462.0,
"completions/mean_length": 294.71875,
"completions/mean_terminated_length": 287.70965576171875,
"completions/min_length": 77.0,
"completions/min_terminated_length": 77.0,
"epoch": 0.7888,
"grad_norm": 12.841826438903809,
"kl": 0.88623046875,
"learning_rate": 1e-06,
"loss": 0.1087,
"num_tokens": 12977742.0,
"reward": 5.2115478515625,
"reward_std": 3.0502023696899414,
"rewards/rm_reward_func/mean": 5.2115478515625,
"rewards/rm_reward_func/std": 14.679587364196777,
"step": 986
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 475.0,
"completions/max_terminated_length": 475.0,
"completions/mean_length": 222.34375,
"completions/mean_terminated_length": 222.34375,
"completions/min_length": 78.0,
"completions/min_terminated_length": 78.0,
"epoch": 0.7896,
"grad_norm": 6.139382839202881,
"kl": 1.00048828125,
"learning_rate": 1e-06,
"loss": 0.0219,
"num_tokens": 12987129.0,
"reward": 1.97021484375,
"reward_std": 4.622243881225586,
"rewards/rm_reward_func/mean": 1.97021484375,
"rewards/rm_reward_func/std": 9.068449020385742,
"step": 987
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 506.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 268.15625,
"completions/mean_terminated_length": 268.15625,
"completions/min_length": 46.0,
"completions/min_terminated_length": 46.0,
"epoch": 0.7904,
"grad_norm": 9.928857803344727,
"kl": 1.89697265625,
"learning_rate": 1e-06,
"loss": -0.1183,
"num_tokens": 12999926.0,
"reward": -0.8626174926757812,
"reward_std": 6.4827094078063965,
"rewards/rm_reward_func/mean": -0.8626174926757812,
"rewards/rm_reward_func/std": 8.545820236206055,
"step": 988
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 446.03125,
"completions/mean_terminated_length": 436.6071472167969,
"completions/min_length": 260.0,
"completions/min_terminated_length": 260.0,
"epoch": 0.7912,
"grad_norm": 8.559710502624512,
"kl": 2.8173828125,
"learning_rate": 1e-06,
"loss": 0.1298,
"num_tokens": 13017151.0,
"reward": -2.3729248046875,
"reward_std": 6.979788303375244,
"rewards/rm_reward_func/mean": -2.3729248046875,
"rewards/rm_reward_func/std": 10.80506706237793,
"step": 989
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 488.0,
"completions/max_terminated_length": 488.0,
"completions/mean_length": 360.53125,
"completions/mean_terminated_length": 360.53125,
"completions/min_length": 228.0,
"completions/min_terminated_length": 228.0,
"epoch": 0.792,
"grad_norm": 14.2783784866333,
"kl": 4.57421875,
"learning_rate": 1e-06,
"loss": 0.2056,
"num_tokens": 13031040.0,
"reward": 3.993865966796875,
"reward_std": 10.918571472167969,
"rewards/rm_reward_func/mean": 3.993865966796875,
"rewards/rm_reward_func/std": 11.101519584655762,
"step": 990
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 418.0,
"completions/mean_length": 253.21875,
"completions/mean_terminated_length": 205.29629516601562,
"completions/min_length": 74.0,
"completions/min_terminated_length": 74.0,
"epoch": 0.7928,
"grad_norm": 12.58895492553711,
"kl": 4.3525390625,
"learning_rate": 1e-06,
"loss": 0.1975,
"num_tokens": 13041079.0,
"reward": -6.41998291015625,
"reward_std": 8.907421112060547,
"rewards/rm_reward_func/mean": -6.41998291015625,
"rewards/rm_reward_func/std": 10.190556526184082,
"step": 991
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 399.0,
"completions/max_terminated_length": 399.0,
"completions/mean_length": 209.8125,
"completions/mean_terminated_length": 209.8125,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.7936,
"grad_norm": 8.097042083740234,
"kl": 4.4404296875,
"learning_rate": 1e-06,
"loss": 0.1805,
"num_tokens": 13053145.0,
"reward": 5.662109375,
"reward_std": 10.280261993408203,
"rewards/rm_reward_func/mean": 5.662109375,
"rewards/rm_reward_func/std": 12.116209030151367,
"step": 992
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 193.875,
"completions/mean_terminated_length": 172.6666717529297,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.7944,
"grad_norm": 16.431880950927734,
"kl": 3.3447265625,
"learning_rate": 1e-06,
"loss": 0.1053,
"num_tokens": 13064117.0,
"reward": 2.54443359375,
"reward_std": 3.539630889892578,
"rewards/rm_reward_func/mean": 2.54443359375,
"rewards/rm_reward_func/std": 13.015660285949707,
"step": 993
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 496.0,
"completions/mean_length": 325.3125,
"completions/mean_terminated_length": 240.45455932617188,
"completions/min_length": 79.0,
"completions/min_terminated_length": 79.0,
"epoch": 0.7952,
"grad_norm": 5.3214592933654785,
"kl": 2.08642578125,
"learning_rate": 1e-06,
"loss": 0.1091,
"num_tokens": 13080759.0,
"reward": 5.818359375,
"reward_std": 6.74766731262207,
"rewards/rm_reward_func/mean": 5.818359375,
"rewards/rm_reward_func/std": 13.01656723022461,
"step": 994
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 440.0,
"completions/mean_length": 398.0625,
"completions/mean_terminated_length": 360.0833435058594,
"completions/min_length": 114.0,
"completions/min_terminated_length": 114.0,
"epoch": 0.796,
"grad_norm": 7.939972400665283,
"kl": 3.1787109375,
"learning_rate": 1e-06,
"loss": 0.0815,
"num_tokens": 13096265.0,
"reward": -3.901641845703125,
"reward_std": 7.492610931396484,
"rewards/rm_reward_func/mean": -3.901641845703125,
"rewards/rm_reward_func/std": 12.713387489318848,
"step": 995
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 433.0,
"completions/max_terminated_length": 433.0,
"completions/mean_length": 216.65625,
"completions/mean_terminated_length": 216.65625,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.7968,
"grad_norm": 17.098487854003906,
"kl": 3.54638671875,
"learning_rate": 1e-06,
"loss": 0.0526,
"num_tokens": 13109534.0,
"reward": -1.665863037109375,
"reward_std": 4.5015339851379395,
"rewards/rm_reward_func/mean": -1.665863037109375,
"rewards/rm_reward_func/std": 8.465290069580078,
"step": 996
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 502.0,
"completions/mean_length": 354.09375,
"completions/mean_terminated_length": 317.65386962890625,
"completions/min_length": 51.0,
"completions/min_terminated_length": 51.0,
"epoch": 0.7976,
"grad_norm": 9.26326847076416,
"kl": 0.84375,
"learning_rate": 1e-06,
"loss": 0.0522,
"num_tokens": 13124065.0,
"reward": 9.808670043945312,
"reward_std": 6.167551040649414,
"rewards/rm_reward_func/mean": 9.808670043945312,
"rewards/rm_reward_func/std": 7.601739406585693,
"step": 997
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 466.0,
"completions/mean_length": 373.6875,
"completions/mean_terminated_length": 364.4666748046875,
"completions/min_length": 240.0,
"completions/min_terminated_length": 240.0,
"epoch": 0.7984,
"grad_norm": 9.506736755371094,
"kl": 7.11279296875,
"learning_rate": 1e-06,
"loss": 0.3447,
"num_tokens": 13138111.0,
"reward": -5.72540283203125,
"reward_std": 7.122351169586182,
"rewards/rm_reward_func/mean": -5.72540283203125,
"rewards/rm_reward_func/std": 7.666444301605225,
"step": 998
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 406.0,
"completions/mean_length": 242.625,
"completions/mean_terminated_length": 167.1999969482422,
"completions/min_length": 78.0,
"completions/min_terminated_length": 78.0,
"epoch": 0.7992,
"grad_norm": 11.145411491394043,
"kl": 1.321533203125,
"learning_rate": 1e-06,
"loss": 0.0769,
"num_tokens": 13149155.0,
"reward": 4.50341796875,
"reward_std": 3.2994678020477295,
"rewards/rm_reward_func/mean": 4.50341796875,
"rewards/rm_reward_func/std": 8.315497398376465,
"step": 999
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.59375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 479.65625,
"completions/mean_terminated_length": 432.3846435546875,
"completions/min_length": 278.0,
"completions/min_terminated_length": 278.0,
"epoch": 0.8,
"grad_norm": 10.525664329528809,
"kl": 3.92578125,
"learning_rate": 1e-06,
"loss": 0.121,
"num_tokens": 13169528.0,
"reward": -4.1995849609375,
"reward_std": 10.804488182067871,
"rewards/rm_reward_func/mean": -4.1995849609375,
"rewards/rm_reward_func/std": 12.529898643493652,
"step": 1000
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 489.0,
"completions/mean_length": 345.59375,
"completions/mean_terminated_length": 269.9545593261719,
"completions/min_length": 84.0,
"completions/min_terminated_length": 84.0,
"epoch": 0.8008,
"grad_norm": 8.35184383392334,
"kl": 2.693359375,
"learning_rate": 1e-06,
"loss": 0.0605,
"num_tokens": 13182515.0,
"reward": 0.7723388671875,
"reward_std": 5.575720310211182,
"rewards/rm_reward_func/mean": 0.7723388671875,
"rewards/rm_reward_func/std": 9.116813659667969,
"step": 1001
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 476.0,
"completions/mean_length": 395.625,
"completions/mean_terminated_length": 379.0000305175781,
"completions/min_length": 204.0,
"completions/min_terminated_length": 204.0,
"epoch": 0.8016,
"grad_norm": 9.205164909362793,
"kl": 3.8544921875,
"learning_rate": 1e-06,
"loss": 0.1071,
"num_tokens": 13197319.0,
"reward": 12.017333984375,
"reward_std": 12.300132751464844,
"rewards/rm_reward_func/mean": 12.017333984375,
"rewards/rm_reward_func/std": 18.185022354125977,
"step": 1002
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 504.0,
"completions/mean_length": 391.875,
"completions/mean_terminated_length": 344.86956787109375,
"completions/min_length": 148.0,
"completions/min_terminated_length": 148.0,
"epoch": 0.8024,
"grad_norm": 7.878791332244873,
"kl": 2.18896484375,
"learning_rate": 1e-06,
"loss": 0.0596,
"num_tokens": 13212147.0,
"reward": -5.649444580078125,
"reward_std": 5.377230167388916,
"rewards/rm_reward_func/mean": -5.649444580078125,
"rewards/rm_reward_func/std": 7.007201671600342,
"step": 1003
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 304.0,
"completions/max_terminated_length": 304.0,
"completions/mean_length": 119.34375,
"completions/mean_terminated_length": 119.34375,
"completions/min_length": 79.0,
"completions/min_terminated_length": 79.0,
"epoch": 0.8032,
"grad_norm": 6.185940742492676,
"kl": 1.81640625,
"learning_rate": 1e-06,
"loss": 0.068,
"num_tokens": 13220862.0,
"reward": 4.4765625,
"reward_std": 0.8225632309913635,
"rewards/rm_reward_func/mean": 4.4765625,
"rewards/rm_reward_func/std": 10.800799369812012,
"step": 1004
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 359.25,
"completions/mean_terminated_length": 299.478271484375,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.804,
"grad_norm": 38.11021041870117,
"kl": 4.56787109375,
"learning_rate": 1e-06,
"loss": 0.0942,
"num_tokens": 13235742.0,
"reward": -5.1068115234375,
"reward_std": 9.743478775024414,
"rewards/rm_reward_func/mean": -5.1068115234375,
"rewards/rm_reward_func/std": 13.0623197555542,
"step": 1005
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 473.0,
"completions/mean_length": 213.75,
"completions/mean_terminated_length": 204.1290283203125,
"completions/min_length": 79.0,
"completions/min_terminated_length": 79.0,
"epoch": 0.8048,
"grad_norm": 13.769661903381348,
"kl": 1.0166015625,
"learning_rate": 1e-06,
"loss": -0.0556,
"num_tokens": 13245302.0,
"reward": -6.537109375,
"reward_std": 4.325830459594727,
"rewards/rm_reward_func/mean": -6.537109375,
"rewards/rm_reward_func/std": 9.237016677856445,
"step": 1006
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 281.875,
"completions/mean_terminated_length": 266.5333557128906,
"completions/min_length": 84.0,
"completions/min_terminated_length": 84.0,
"epoch": 0.8056,
"grad_norm": 13.841572761535645,
"kl": 5.92578125,
"learning_rate": 1e-06,
"loss": 0.1448,
"num_tokens": 13256546.0,
"reward": -1.900146484375,
"reward_std": 8.712725639343262,
"rewards/rm_reward_func/mean": -1.900146484375,
"rewards/rm_reward_func/std": 12.52529525756836,
"step": 1007
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 501.0,
"completions/max_terminated_length": 501.0,
"completions/mean_length": 309.78125,
"completions/mean_terminated_length": 309.78125,
"completions/min_length": 23.0,
"completions/min_terminated_length": 23.0,
"epoch": 0.8064,
"grad_norm": 21.372207641601562,
"kl": 4.873291015625,
"learning_rate": 1e-06,
"loss": 0.0049,
"num_tokens": 13268731.0,
"reward": 5.0888671875,
"reward_std": 5.007624626159668,
"rewards/rm_reward_func/mean": 5.0888671875,
"rewards/rm_reward_func/std": 22.30028533935547,
"step": 1008
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 496.0,
"completions/mean_length": 314.96875,
"completions/mean_terminated_length": 286.8214416503906,
"completions/min_length": 44.0,
"completions/min_terminated_length": 44.0,
"epoch": 0.8072,
"grad_norm": 25.30167579650879,
"kl": 9.71826171875,
"learning_rate": 1e-06,
"loss": 0.4105,
"num_tokens": 13285794.0,
"reward": -8.27099609375,
"reward_std": 7.645841598510742,
"rewards/rm_reward_func/mean": -8.27099609375,
"rewards/rm_reward_func/std": 16.137731552124023,
"step": 1009
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 410.0,
"completions/mean_length": 273.875,
"completions/mean_terminated_length": 266.19354248046875,
"completions/min_length": 72.0,
"completions/min_terminated_length": 72.0,
"epoch": 0.808,
"grad_norm": 8.614705085754395,
"kl": 0.37060546875,
"learning_rate": 1e-06,
"loss": -0.0065,
"num_tokens": 13299302.0,
"reward": 20.390045166015625,
"reward_std": 5.194117546081543,
"rewards/rm_reward_func/mean": 20.390045166015625,
"rewards/rm_reward_func/std": 17.19424057006836,
"step": 1010
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 504.0,
"completions/mean_length": 342.6875,
"completions/mean_terminated_length": 303.6153869628906,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.8088,
"grad_norm": 4.921807289123535,
"kl": 1.82275390625,
"learning_rate": 1e-06,
"loss": 0.0481,
"num_tokens": 13313140.0,
"reward": 4.9627838134765625,
"reward_std": 6.066807270050049,
"rewards/rm_reward_func/mean": 4.9627838134765625,
"rewards/rm_reward_func/std": 9.324835777282715,
"step": 1011
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 455.0,
"completions/mean_length": 245.84375,
"completions/mean_terminated_length": 171.3199920654297,
"completions/min_length": 65.0,
"completions/min_terminated_length": 65.0,
"epoch": 0.8096,
"grad_norm": 19.971662521362305,
"kl": 3.2412109375,
"learning_rate": 1e-06,
"loss": 0.0408,
"num_tokens": 13325503.0,
"reward": -3.34765625,
"reward_std": 4.555385589599609,
"rewards/rm_reward_func/mean": -3.34765625,
"rewards/rm_reward_func/std": 15.34835433959961,
"step": 1012
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 365.0,
"completions/max_terminated_length": 365.0,
"completions/mean_length": 193.03125,
"completions/mean_terminated_length": 193.03125,
"completions/min_length": 79.0,
"completions/min_terminated_length": 79.0,
"epoch": 0.8104,
"grad_norm": 11.546648025512695,
"kl": 1.3896484375,
"learning_rate": 1e-06,
"loss": 0.0518,
"num_tokens": 13334680.0,
"reward": -4.6224365234375,
"reward_std": 3.3871943950653076,
"rewards/rm_reward_func/mean": -4.6224365234375,
"rewards/rm_reward_func/std": 9.441699981689453,
"step": 1013
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 431.0,
"completions/max_terminated_length": 431.0,
"completions/mean_length": 322.84375,
"completions/mean_terminated_length": 322.84375,
"completions/min_length": 155.0,
"completions/min_terminated_length": 155.0,
"epoch": 0.8112,
"grad_norm": 6.9250359535217285,
"kl": 1.911376953125,
"learning_rate": 1e-06,
"loss": 0.0922,
"num_tokens": 13347131.0,
"reward": 4.8478851318359375,
"reward_std": 8.108898162841797,
"rewards/rm_reward_func/mean": 4.8478851318359375,
"rewards/rm_reward_func/std": 15.027045249938965,
"step": 1014
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 500.0,
"completions/mean_length": 232.625,
"completions/mean_terminated_length": 192.71429443359375,
"completions/min_length": 52.0,
"completions/min_terminated_length": 52.0,
"epoch": 0.812,
"grad_norm": 35.626102447509766,
"kl": 0.970703125,
"learning_rate": 1e-06,
"loss": 0.0813,
"num_tokens": 13360111.0,
"reward": 0.5765380859375,
"reward_std": 6.022714614868164,
"rewards/rm_reward_func/mean": 0.5765380859375,
"rewards/rm_reward_func/std": 14.149896621704102,
"step": 1015
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 380.0,
"completions/mean_length": 352.46875,
"completions/mean_terminated_length": 299.29168701171875,
"completions/min_length": 220.0,
"completions/min_terminated_length": 220.0,
"epoch": 0.8128,
"grad_norm": 9.04706859588623,
"kl": 0.98681640625,
"learning_rate": 1e-06,
"loss": 0.0161,
"num_tokens": 13373950.0,
"reward": -5.08154296875,
"reward_std": 3.3396553993225098,
"rewards/rm_reward_func/mean": -5.08154296875,
"rewards/rm_reward_func/std": 7.902137756347656,
"step": 1016
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 414.5,
"completions/mean_terminated_length": 392.0,
"completions/min_length": 206.0,
"completions/min_terminated_length": 206.0,
"epoch": 0.8136,
"grad_norm": 7.317805290222168,
"kl": 0.775390625,
"learning_rate": 1e-06,
"loss": -0.0302,
"num_tokens": 13391822.0,
"reward": 5.2294158935546875,
"reward_std": 6.866477012634277,
"rewards/rm_reward_func/mean": 5.2294158935546875,
"rewards/rm_reward_func/std": 10.840312957763672,
"step": 1017
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 507.0,
"completions/mean_length": 267.6875,
"completions/mean_terminated_length": 222.44444274902344,
"completions/min_length": 78.0,
"completions/min_terminated_length": 78.0,
"epoch": 0.8144,
"grad_norm": 8.47047233581543,
"kl": 0.93603515625,
"learning_rate": 1e-06,
"loss": 0.0759,
"num_tokens": 13402404.0,
"reward": 0.318359375,
"reward_std": 2.553496837615967,
"rewards/rm_reward_func/mean": 0.318359375,
"rewards/rm_reward_func/std": 9.620682716369629,
"step": 1018
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 509.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 298.75,
"completions/mean_terminated_length": 298.75,
"completions/min_length": 91.0,
"completions/min_terminated_length": 91.0,
"epoch": 0.8152,
"grad_norm": 10.068121910095215,
"kl": 0.60400390625,
"learning_rate": 1e-06,
"loss": 0.0192,
"num_tokens": 13414804.0,
"reward": -0.441314697265625,
"reward_std": 2.4171152114868164,
"rewards/rm_reward_func/mean": -0.441314697265625,
"rewards/rm_reward_func/std": 6.881007671356201,
"step": 1019
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 498.0,
"completions/max_terminated_length": 498.0,
"completions/mean_length": 343.5,
"completions/mean_terminated_length": 343.5,
"completions/min_length": 64.0,
"completions/min_terminated_length": 64.0,
"epoch": 0.816,
"grad_norm": 10.040746688842773,
"kl": 3.17138671875,
"learning_rate": 1e-06,
"loss": 0.005,
"num_tokens": 13428028.0,
"reward": 8.162225723266602,
"reward_std": 8.07081413269043,
"rewards/rm_reward_func/mean": 8.162225723266602,
"rewards/rm_reward_func/std": 20.003002166748047,
"step": 1020
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 482.0,
"completions/mean_length": 391.78125,
"completions/mean_terminated_length": 351.7083435058594,
"completions/min_length": 187.0,
"completions/min_terminated_length": 187.0,
"epoch": 0.8168,
"grad_norm": 10.024792671203613,
"kl": 2.970703125,
"learning_rate": 1e-06,
"loss": 0.0447,
"num_tokens": 13443005.0,
"reward": -0.19919586181640625,
"reward_std": 7.3775715827941895,
"rewards/rm_reward_func/mean": -0.19919586181640625,
"rewards/rm_reward_func/std": 14.476346015930176,
"step": 1021
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 340.09375,
"completions/mean_terminated_length": 334.5483703613281,
"completions/min_length": 79.0,
"completions/min_terminated_length": 79.0,
"epoch": 0.8176,
"grad_norm": 7.288039684295654,
"kl": 0.75927734375,
"learning_rate": 1e-06,
"loss": 0.001,
"num_tokens": 13456536.0,
"reward": 9.702880859375,
"reward_std": 5.229021072387695,
"rewards/rm_reward_func/mean": 9.702880859375,
"rewards/rm_reward_func/std": 9.884917259216309,
"step": 1022
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 433.0,
"completions/max_terminated_length": 433.0,
"completions/mean_length": 216.71875,
"completions/mean_terminated_length": 216.71875,
"completions/min_length": 79.0,
"completions/min_terminated_length": 79.0,
"epoch": 0.8184,
"grad_norm": 5.475179672241211,
"kl": 0.51953125,
"learning_rate": 1e-06,
"loss": 0.0019,
"num_tokens": 13466887.0,
"reward": 9.95068359375,
"reward_std": 3.0980048179626465,
"rewards/rm_reward_func/mean": 9.95068359375,
"rewards/rm_reward_func/std": 5.0042338371276855,
"step": 1023
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 334.65625,
"completions/mean_terminated_length": 301.8148193359375,
"completions/min_length": 45.0,
"completions/min_terminated_length": 45.0,
"epoch": 0.8192,
"grad_norm": 9.030261039733887,
"kl": 0.4716796875,
"learning_rate": 1e-06,
"loss": -0.0265,
"num_tokens": 13480036.0,
"reward": 2.835418701171875,
"reward_std": 5.5337724685668945,
"rewards/rm_reward_func/mean": 2.835418701171875,
"rewards/rm_reward_func/std": 8.441008567810059,
"step": 1024
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 472.0,
"completions/mean_length": 232.0625,
"completions/mean_terminated_length": 223.03225708007812,
"completions/min_length": 79.0,
"completions/min_terminated_length": 79.0,
"epoch": 0.82,
"grad_norm": 13.575851440429688,
"kl": 1.310546875,
"learning_rate": 1e-06,
"loss": 0.0396,
"num_tokens": 13491534.0,
"reward": 4.1461181640625,
"reward_std": 4.509701728820801,
"rewards/rm_reward_func/mean": 4.1461181640625,
"rewards/rm_reward_func/std": 10.166535377502441,
"step": 1025
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 408.0,
"completions/max_terminated_length": 408.0,
"completions/mean_length": 250.875,
"completions/mean_terminated_length": 250.875,
"completions/min_length": 95.0,
"completions/min_terminated_length": 95.0,
"epoch": 0.8208,
"grad_norm": 10.076627731323242,
"kl": 3.59375,
"learning_rate": 1e-06,
"loss": -0.0269,
"num_tokens": 13505082.0,
"reward": 2.3536376953125,
"reward_std": 10.482784271240234,
"rewards/rm_reward_func/mean": 2.3536376953125,
"rewards/rm_reward_func/std": 24.783512115478516,
"step": 1026
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 411.0,
"completions/max_terminated_length": 411.0,
"completions/mean_length": 260.21875,
"completions/mean_terminated_length": 260.21875,
"completions/min_length": 59.0,
"completions/min_terminated_length": 59.0,
"epoch": 0.8216,
"grad_norm": 27.619287490844727,
"kl": 2.212890625,
"learning_rate": 1e-06,
"loss": 0.0159,
"num_tokens": 13515673.0,
"reward": 0.1591796875,
"reward_std": 5.797090530395508,
"rewards/rm_reward_func/mean": 0.1591796875,
"rewards/rm_reward_func/std": 17.642372131347656,
"step": 1027
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 446.0,
"completions/mean_length": 341.03125,
"completions/mean_terminated_length": 316.6071472167969,
"completions/min_length": 154.0,
"completions/min_terminated_length": 154.0,
"epoch": 0.8224,
"grad_norm": 8.583765029907227,
"kl": 2.4423828125,
"learning_rate": 1e-06,
"loss": 0.072,
"num_tokens": 13531434.0,
"reward": 7.218017578125,
"reward_std": 3.8263235092163086,
"rewards/rm_reward_func/mean": 7.218017578125,
"rewards/rm_reward_func/std": 22.30869483947754,
"step": 1028
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 504.0,
"completions/mean_length": 301.0,
"completions/mean_terminated_length": 294.19354248046875,
"completions/min_length": 79.0,
"completions/min_terminated_length": 79.0,
"epoch": 0.8232,
"grad_norm": 17.819374084472656,
"kl": 0.42431640625,
"learning_rate": 1e-06,
"loss": 0.0075,
"num_tokens": 13543170.0,
"reward": 13.765625,
"reward_std": 3.121476173400879,
"rewards/rm_reward_func/mean": 13.765625,
"rewards/rm_reward_func/std": 17.44659996032715,
"step": 1029
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 463.0,
"completions/mean_length": 335.4375,
"completions/mean_terminated_length": 255.18182373046875,
"completions/min_length": 79.0,
"completions/min_terminated_length": 79.0,
"epoch": 0.824,
"grad_norm": 12.179718971252441,
"kl": 1.9833984375,
"learning_rate": 1e-06,
"loss": 0.048,
"num_tokens": 13558328.0,
"reward": -1.099609375,
"reward_std": 2.019794464111328,
"rewards/rm_reward_func/mean": -1.099609375,
"rewards/rm_reward_func/std": 11.247794151306152,
"step": 1030
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 479.0,
"completions/mean_length": 319.6875,
"completions/mean_terminated_length": 255.58334350585938,
"completions/min_length": 79.0,
"completions/min_terminated_length": 79.0,
"epoch": 0.8248,
"grad_norm": 5.296995162963867,
"kl": 0.63330078125,
"learning_rate": 1e-06,
"loss": -0.0521,
"num_tokens": 13571750.0,
"reward": 8.6865234375,
"reward_std": 5.767685890197754,
"rewards/rm_reward_func/mean": 8.6865234375,
"rewards/rm_reward_func/std": 10.389171600341797,
"step": 1031
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 483.0,
"completions/mean_length": 244.0625,
"completions/mean_terminated_length": 216.34483337402344,
"completions/min_length": 65.0,
"completions/min_terminated_length": 65.0,
"epoch": 0.8256,
"grad_norm": 8.053586959838867,
"kl": 2.3212890625,
"learning_rate": 1e-06,
"loss": 0.034,
"num_tokens": 13581456.0,
"reward": -4.644775390625,
"reward_std": 6.1426496505737305,
"rewards/rm_reward_func/mean": -4.644775390625,
"rewards/rm_reward_func/std": 9.984515190124512,
"step": 1032
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 485.0,
"completions/mean_length": 285.1875,
"completions/mean_terminated_length": 261.7241516113281,
"completions/min_length": 79.0,
"completions/min_terminated_length": 79.0,
"epoch": 0.8264,
"grad_norm": 6.782697677612305,
"kl": 0.65087890625,
"learning_rate": 1e-06,
"loss": 0.0033,
"num_tokens": 13593006.0,
"reward": 0.996185302734375,
"reward_std": 2.4062700271606445,
"rewards/rm_reward_func/mean": 0.996185302734375,
"rewards/rm_reward_func/std": 5.100530624389648,
"step": 1033
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 303.96875,
"completions/mean_terminated_length": 297.258056640625,
"completions/min_length": 118.0,
"completions/min_terminated_length": 118.0,
"epoch": 0.8272,
"grad_norm": 11.949031829833984,
"kl": 3.359375,
"learning_rate": 1e-06,
"loss": 0.0168,
"num_tokens": 13604669.0,
"reward": -4.284423828125,
"reward_std": 13.229473114013672,
"rewards/rm_reward_func/mean": -4.284423828125,
"rewards/rm_reward_func/std": 17.82453727722168,
"step": 1034
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 431.15625,
"completions/mean_terminated_length": 404.2083435058594,
"completions/min_length": 264.0,
"completions/min_terminated_length": 264.0,
"epoch": 0.828,
"grad_norm": 6.652060031890869,
"kl": 1.0,
"learning_rate": 1e-06,
"loss": 0.0325,
"num_tokens": 13620410.0,
"reward": 1.5401611328125,
"reward_std": 7.661131858825684,
"rewards/rm_reward_func/mean": 1.5401611328125,
"rewards/rm_reward_func/std": 9.722722053527832,
"step": 1035
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 497.0,
"completions/mean_length": 262.53125,
"completions/mean_terminated_length": 226.8928680419922,
"completions/min_length": 79.0,
"completions/min_terminated_length": 79.0,
"epoch": 0.8288,
"grad_norm": 12.548855781555176,
"kl": 3.18408203125,
"learning_rate": 1e-06,
"loss": 0.1386,
"num_tokens": 13632595.0,
"reward": 0.58056640625,
"reward_std": 2.6908631324768066,
"rewards/rm_reward_func/mean": 0.58056640625,
"rewards/rm_reward_func/std": 9.962352752685547,
"step": 1036
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 465.0,
"completions/mean_length": 333.3125,
"completions/mean_terminated_length": 327.5483703613281,
"completions/min_length": 129.0,
"completions/min_terminated_length": 129.0,
"epoch": 0.8296,
"grad_norm": 7.802600383758545,
"kl": 3.310546875,
"learning_rate": 1e-06,
"loss": 0.081,
"num_tokens": 13648397.0,
"reward": -1.08026123046875,
"reward_std": 7.778307914733887,
"rewards/rm_reward_func/mean": -1.08026123046875,
"rewards/rm_reward_func/std": 9.398695945739746,
"step": 1037
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 468.0,
"completions/max_terminated_length": 468.0,
"completions/mean_length": 261.5,
"completions/mean_terminated_length": 261.5,
"completions/min_length": 118.0,
"completions/min_terminated_length": 118.0,
"epoch": 0.8304,
"grad_norm": 13.804716110229492,
"kl": 1.47998046875,
"learning_rate": 1e-06,
"loss": 0.0421,
"num_tokens": 13659029.0,
"reward": -1.4334716796875,
"reward_std": 3.4379241466522217,
"rewards/rm_reward_func/mean": -1.4334716796875,
"rewards/rm_reward_func/std": 13.476388931274414,
"step": 1038
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 507.0,
"completions/mean_length": 397.59375,
"completions/mean_terminated_length": 376.40740966796875,
"completions/min_length": 129.0,
"completions/min_terminated_length": 129.0,
"epoch": 0.8312,
"grad_norm": 11.404579162597656,
"kl": 1.0322265625,
"learning_rate": 1e-06,
"loss": -0.0507,
"num_tokens": 13675640.0,
"reward": 4.508544921875,
"reward_std": 11.643041610717773,
"rewards/rm_reward_func/mean": 4.508544921875,
"rewards/rm_reward_func/std": 12.394889831542969,
"step": 1039
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 505.0,
"completions/mean_length": 358.3125,
"completions/mean_terminated_length": 315.2799987792969,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.832,
"grad_norm": 7.375335216522217,
"kl": 1.384765625,
"learning_rate": 1e-06,
"loss": 0.0147,
"num_tokens": 13689202.0,
"reward": 0.5595703125,
"reward_std": 5.928117752075195,
"rewards/rm_reward_func/mean": 0.5595703125,
"rewards/rm_reward_func/std": 8.306135177612305,
"step": 1040
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 498.0,
"completions/mean_length": 442.28125,
"completions/mean_terminated_length": 400.45001220703125,
"completions/min_length": 245.0,
"completions/min_terminated_length": 245.0,
"epoch": 0.8328,
"grad_norm": 103.06358337402344,
"kl": 5.14306640625,
"learning_rate": 1e-06,
"loss": 0.2078,
"num_tokens": 13706795.0,
"reward": -7.9290771484375,
"reward_std": 5.579919815063477,
"rewards/rm_reward_func/mean": -7.9290771484375,
"rewards/rm_reward_func/std": 9.49826717376709,
"step": 1041
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 460.0,
"completions/max_terminated_length": 460.0,
"completions/mean_length": 267.0625,
"completions/mean_terminated_length": 267.0625,
"completions/min_length": 43.0,
"completions/min_terminated_length": 43.0,
"epoch": 0.8336,
"grad_norm": 13.862227439880371,
"kl": 2.5947265625,
"learning_rate": 1e-06,
"loss": -0.0647,
"num_tokens": 13717565.0,
"reward": -8.9283447265625,
"reward_std": 2.9016993045806885,
"rewards/rm_reward_func/mean": -8.9283447265625,
"rewards/rm_reward_func/std": 7.622008800506592,
"step": 1042
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 498.0,
"completions/mean_length": 388.78125,
"completions/mean_terminated_length": 376.03448486328125,
"completions/min_length": 245.0,
"completions/min_terminated_length": 245.0,
"epoch": 0.8344,
"grad_norm": 10.778160095214844,
"kl": 1.15380859375,
"learning_rate": 1e-06,
"loss": 0.0711,
"num_tokens": 13735566.0,
"reward": 7.612060546875,
"reward_std": 7.548920631408691,
"rewards/rm_reward_func/mean": 7.612060546875,
"rewards/rm_reward_func/std": 13.313557624816895,
"step": 1043
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 371.40625,
"completions/mean_terminated_length": 366.8709716796875,
"completions/min_length": 113.0,
"completions/min_terminated_length": 113.0,
"epoch": 0.8352,
"grad_norm": 8.611907958984375,
"kl": 3.005126953125,
"learning_rate": 1e-06,
"loss": 0.0158,
"num_tokens": 13749723.0,
"reward": 3.6009521484375,
"reward_std": 10.527230262756348,
"rewards/rm_reward_func/mean": 3.6009521484375,
"rewards/rm_reward_func/std": 13.915834426879883,
"step": 1044
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 301.0,
"completions/mean_terminated_length": 279.17242431640625,
"completions/min_length": 47.0,
"completions/min_terminated_length": 47.0,
"epoch": 0.836,
"grad_norm": 19.481334686279297,
"kl": 6.16796875,
"learning_rate": 1e-06,
"loss": 0.1613,
"num_tokens": 13765139.0,
"reward": -4.6524658203125,
"reward_std": 3.753931999206543,
"rewards/rm_reward_func/mean": -4.6524658203125,
"rewards/rm_reward_func/std": 11.036279678344727,
"step": 1045
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 481.0,
"completions/mean_length": 272.84375,
"completions/mean_terminated_length": 228.55555725097656,
"completions/min_length": 66.0,
"completions/min_terminated_length": 66.0,
"epoch": 0.8368,
"grad_norm": 18.560142517089844,
"kl": 7.0703125,
"learning_rate": 1e-06,
"loss": 0.2253,
"num_tokens": 13776190.0,
"reward": -9.101318359375,
"reward_std": 8.66612434387207,
"rewards/rm_reward_func/mean": -9.101318359375,
"rewards/rm_reward_func/std": 10.977357864379883,
"step": 1046
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 369.15625,
"completions/mean_terminated_length": 329.1600036621094,
"completions/min_length": 95.0,
"completions/min_terminated_length": 95.0,
"epoch": 0.8376,
"grad_norm": 14.926949501037598,
"kl": 0.888671875,
"learning_rate": 1e-06,
"loss": -0.0087,
"num_tokens": 13789795.0,
"reward": 0.6767578125,
"reward_std": 8.464103698730469,
"rewards/rm_reward_func/mean": 0.6767578125,
"rewards/rm_reward_func/std": 14.696283340454102,
"step": 1047
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 487.0,
"completions/mean_length": 358.0,
"completions/mean_terminated_length": 336.0,
"completions/min_length": 168.0,
"completions/min_terminated_length": 168.0,
"epoch": 0.8384,
"grad_norm": 14.153160095214844,
"kl": 3.96533203125,
"learning_rate": 1e-06,
"loss": 0.0663,
"num_tokens": 13805603.0,
"reward": -7.618194580078125,
"reward_std": 5.023479461669922,
"rewards/rm_reward_func/mean": -7.618194580078125,
"rewards/rm_reward_func/std": 13.050457000732422,
"step": 1048
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 490.0,
"completions/mean_length": 377.9375,
"completions/mean_terminated_length": 369.0000305175781,
"completions/min_length": 66.0,
"completions/min_terminated_length": 66.0,
"epoch": 0.8392,
"grad_norm": 9.757351875305176,
"kl": 2.5224609375,
"learning_rate": 1e-06,
"loss": 0.0443,
"num_tokens": 13822233.0,
"reward": 3.964599609375,
"reward_std": 10.471548080444336,
"rewards/rm_reward_func/mean": 3.964599609375,
"rewards/rm_reward_func/std": 16.459672927856445,
"step": 1049
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 476.0,
"completions/mean_length": 310.1875,
"completions/mean_terminated_length": 281.3571472167969,
"completions/min_length": 29.0,
"completions/min_terminated_length": 29.0,
"epoch": 0.84,
"grad_norm": 12.101102828979492,
"kl": 3.15673828125,
"learning_rate": 1e-06,
"loss": -0.0473,
"num_tokens": 13833951.0,
"reward": -2.974365234375,
"reward_std": 9.494062423706055,
"rewards/rm_reward_func/mean": -2.974365234375,
"rewards/rm_reward_func/std": 17.267580032348633,
"step": 1050
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 234.0625,
"completions/mean_terminated_length": 194.35714721679688,
"completions/min_length": 79.0,
"completions/min_terminated_length": 79.0,
"epoch": 0.8408,
"grad_norm": 6.150353908538818,
"kl": 2.26953125,
"learning_rate": 1e-06,
"loss": 0.0698,
"num_tokens": 13844073.0,
"reward": -1.613037109375,
"reward_std": 4.632699489593506,
"rewards/rm_reward_func/mean": -1.613037109375,
"rewards/rm_reward_func/std": 8.01832103729248,
"step": 1051
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 359.6875,
"completions/mean_terminated_length": 317.03997802734375,
"completions/min_length": 79.0,
"completions/min_terminated_length": 79.0,
"epoch": 0.8416,
"grad_norm": 9.147581100463867,
"kl": 3.4599609375,
"learning_rate": 1e-06,
"loss": 0.1188,
"num_tokens": 13858791.0,
"reward": -1.190673828125,
"reward_std": 5.757209300994873,
"rewards/rm_reward_func/mean": -1.190673828125,
"rewards/rm_reward_func/std": 9.849881172180176,
"step": 1052
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 496.0,
"completions/mean_length": 333.71875,
"completions/mean_terminated_length": 263.9565124511719,
"completions/min_length": 98.0,
"completions/min_terminated_length": 98.0,
"epoch": 0.8424,
"grad_norm": 12.606836318969727,
"kl": 3.59375,
"learning_rate": 1e-06,
"loss": 0.1426,
"num_tokens": 13873742.0,
"reward": -8.640029907226562,
"reward_std": 7.198362350463867,
"rewards/rm_reward_func/mean": -8.640029907226562,
"rewards/rm_reward_func/std": 10.380491256713867,
"step": 1053
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 469.0,
"completions/mean_length": 230.84375,
"completions/mean_terminated_length": 221.77418518066406,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.8432,
"grad_norm": 6.97705078125,
"kl": 2.47314453125,
"learning_rate": 1e-06,
"loss": 0.173,
"num_tokens": 13883497.0,
"reward": -2.553131103515625,
"reward_std": 3.148016929626465,
"rewards/rm_reward_func/mean": -2.553131103515625,
"rewards/rm_reward_func/std": 8.437515258789062,
"step": 1054
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 483.0,
"completions/mean_length": 325.25,
"completions/mean_terminated_length": 213.1999969482422,
"completions/min_length": 79.0,
"completions/min_terminated_length": 79.0,
"epoch": 0.844,
"grad_norm": 7.593416213989258,
"kl": 0.8896484375,
"learning_rate": 1e-06,
"loss": 0.0335,
"num_tokens": 13897529.0,
"reward": 6.3282470703125,
"reward_std": 3.7554547786712646,
"rewards/rm_reward_func/mean": 6.3282470703125,
"rewards/rm_reward_func/std": 13.18265438079834,
"step": 1055
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 338.25,
"completions/mean_terminated_length": 289.6000061035156,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.8448,
"grad_norm": 7.9972734451293945,
"kl": 1.67822265625,
"learning_rate": 1e-06,
"loss": 0.0365,
"num_tokens": 13910961.0,
"reward": 9.41973876953125,
"reward_std": 5.808597564697266,
"rewards/rm_reward_func/mean": 9.41973876953125,
"rewards/rm_reward_func/std": 13.108800888061523,
"step": 1056
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 499.0,
"completions/mean_length": 378.15625,
"completions/mean_terminated_length": 317.31817626953125,
"completions/min_length": 63.0,
"completions/min_terminated_length": 63.0,
"epoch": 0.8456,
"grad_norm": 9.818436622619629,
"kl": 2.041015625,
"learning_rate": 1e-06,
"loss": 0.0809,
"num_tokens": 13925710.0,
"reward": 4.3310546875,
"reward_std": 6.725727081298828,
"rewards/rm_reward_func/mean": 4.3310546875,
"rewards/rm_reward_func/std": 8.763275146484375,
"step": 1057
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 465.0,
"completions/mean_length": 377.5,
"completions/mean_terminated_length": 352.59259033203125,
"completions/min_length": 114.0,
"completions/min_terminated_length": 114.0,
"epoch": 0.8464,
"grad_norm": 11.298351287841797,
"kl": 0.86474609375,
"learning_rate": 1e-06,
"loss": 0.1117,
"num_tokens": 13939478.0,
"reward": 4.32666015625,
"reward_std": 8.901585578918457,
"rewards/rm_reward_func/mean": 4.32666015625,
"rewards/rm_reward_func/std": 16.733640670776367,
"step": 1058
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 472.0,
"completions/mean_length": 346.40625,
"completions/mean_terminated_length": 300.0400085449219,
"completions/min_length": 70.0,
"completions/min_terminated_length": 70.0,
"epoch": 0.8472,
"grad_norm": 7.649352550506592,
"kl": 2.932861328125,
"learning_rate": 1e-06,
"loss": -0.004,
"num_tokens": 13955987.0,
"reward": 10.8458251953125,
"reward_std": 10.861635208129883,
"rewards/rm_reward_func/mean": 10.8458251953125,
"rewards/rm_reward_func/std": 20.45244598388672,
"step": 1059
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 251.59375,
"completions/mean_terminated_length": 203.37037658691406,
"completions/min_length": 79.0,
"completions/min_terminated_length": 79.0,
"epoch": 0.848,
"grad_norm": 14.20515251159668,
"kl": 1.2626953125,
"learning_rate": 1e-06,
"loss": -0.1202,
"num_tokens": 13967622.0,
"reward": 3.55889892578125,
"reward_std": 3.865619659423828,
"rewards/rm_reward_func/mean": 3.55889892578125,
"rewards/rm_reward_func/std": 14.522747039794922,
"step": 1060
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 447.0,
"completions/max_terminated_length": 447.0,
"completions/mean_length": 204.1875,
"completions/mean_terminated_length": 204.1875,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.8488,
"grad_norm": 5.686960220336914,
"kl": 0.53759765625,
"learning_rate": 1e-06,
"loss": -0.0044,
"num_tokens": 13976516.0,
"reward": 10.380859375,
"reward_std": 1.672519326210022,
"rewards/rm_reward_func/mean": 10.380859375,
"rewards/rm_reward_func/std": 4.958350658416748,
"step": 1061
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 455.0,
"completions/max_terminated_length": 455.0,
"completions/mean_length": 294.875,
"completions/mean_terminated_length": 294.875,
"completions/min_length": 115.0,
"completions/min_terminated_length": 115.0,
"epoch": 0.8496,
"grad_norm": 6.868638038635254,
"kl": 0.47802734375,
"learning_rate": 1e-06,
"loss": -0.0463,
"num_tokens": 13987656.0,
"reward": -3.4047164916992188,
"reward_std": 2.7297439575195312,
"rewards/rm_reward_func/mean": -3.4047164916992188,
"rewards/rm_reward_func/std": 8.868703842163086,
"step": 1062
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 484.0,
"completions/mean_length": 354.15625,
"completions/mean_terminated_length": 309.9599914550781,
"completions/min_length": 83.0,
"completions/min_terminated_length": 83.0,
"epoch": 0.8504,
"grad_norm": 10.120816230773926,
"kl": 0.4521484375,
"learning_rate": 1e-06,
"loss": 0.0789,
"num_tokens": 14001477.0,
"reward": 10.3330078125,
"reward_std": 5.511585235595703,
"rewards/rm_reward_func/mean": 10.3330078125,
"rewards/rm_reward_func/std": 11.582442283630371,
"step": 1063
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 501.0,
"completions/mean_length": 383.15625,
"completions/mean_terminated_length": 347.0799865722656,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.8512,
"grad_norm": 8.560920715332031,
"kl": 1.580078125,
"learning_rate": 1e-06,
"loss": 0.1501,
"num_tokens": 14016578.0,
"reward": 15.5166015625,
"reward_std": 7.954902648925781,
"rewards/rm_reward_func/mean": 15.5166015625,
"rewards/rm_reward_func/std": 15.997610092163086,
"step": 1064
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 511.0,
"completions/mean_length": 219.71875,
"completions/mean_terminated_length": 177.96429443359375,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.852,
"grad_norm": 9.301816940307617,
"kl": 0.8408203125,
"learning_rate": 1e-06,
"loss": 0.0325,
"num_tokens": 14026161.0,
"reward": 0.931640625,
"reward_std": 2.5424585342407227,
"rewards/rm_reward_func/mean": 0.931640625,
"rewards/rm_reward_func/std": 9.632967948913574,
"step": 1065
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 511.0,
"completions/max_terminated_length": 511.0,
"completions/mean_length": 292.34375,
"completions/mean_terminated_length": 292.34375,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.8528,
"grad_norm": 12.27116584777832,
"kl": 1.8857421875,
"learning_rate": 1e-06,
"loss": -0.0008,
"num_tokens": 14038460.0,
"reward": 5.19140625,
"reward_std": 2.821535110473633,
"rewards/rm_reward_func/mean": 5.19140625,
"rewards/rm_reward_func/std": 11.627610206604004,
"step": 1066
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 503.0,
"completions/mean_length": 349.5625,
"completions/mean_terminated_length": 344.32257080078125,
"completions/min_length": 71.0,
"completions/min_terminated_length": 71.0,
"epoch": 0.8536,
"grad_norm": 11.470945358276367,
"kl": 0.71337890625,
"learning_rate": 1e-06,
"loss": -0.0169,
"num_tokens": 14051606.0,
"reward": 3.7967529296875,
"reward_std": 6.249090194702148,
"rewards/rm_reward_func/mean": 3.7967529296875,
"rewards/rm_reward_func/std": 6.5110979080200195,
"step": 1067
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 500.0,
"completions/mean_length": 312.90625,
"completions/mean_terminated_length": 299.63336181640625,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.8544,
"grad_norm": 10.407036781311035,
"kl": 0.72705078125,
"learning_rate": 1e-06,
"loss": 0.0446,
"num_tokens": 14063891.0,
"reward": 16.3878173828125,
"reward_std": 8.335765838623047,
"rewards/rm_reward_func/mean": 16.3878173828125,
"rewards/rm_reward_func/std": 16.84733009338379,
"step": 1068
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 281.96875,
"completions/mean_terminated_length": 266.63336181640625,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.8552,
"grad_norm": 7.953994274139404,
"kl": 0.6884765625,
"learning_rate": 1e-06,
"loss": 0.0595,
"num_tokens": 14074970.0,
"reward": 3.028076171875,
"reward_std": 4.655251979827881,
"rewards/rm_reward_func/mean": 3.028076171875,
"rewards/rm_reward_func/std": 14.149062156677246,
"step": 1069
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 479.0,
"completions/mean_length": 300.375,
"completions/mean_terminated_length": 261.1851806640625,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.856,
"grad_norm": 5.069062232971191,
"kl": 0.404541015625,
"learning_rate": 1e-06,
"loss": 0.0521,
"num_tokens": 14088094.0,
"reward": 1.788726806640625,
"reward_std": 5.874140739440918,
"rewards/rm_reward_func/mean": 1.788726806640625,
"rewards/rm_reward_func/std": 10.132195472717285,
"step": 1070
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 507.0,
"completions/mean_length": 198.875,
"completions/mean_terminated_length": 188.77418518066406,
"completions/min_length": 54.0,
"completions/min_terminated_length": 54.0,
"epoch": 0.8568,
"grad_norm": 11.527831077575684,
"kl": 1.47314453125,
"learning_rate": 1e-06,
"loss": 0.1007,
"num_tokens": 14097274.0,
"reward": -1.9715576171875,
"reward_std": 2.8827223777770996,
"rewards/rm_reward_func/mean": -1.9715576171875,
"rewards/rm_reward_func/std": 5.209477424621582,
"step": 1071
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 503.0,
"completions/mean_length": 449.09375,
"completions/mean_terminated_length": 411.3500061035156,
"completions/min_length": 336.0,
"completions/min_terminated_length": 336.0,
"epoch": 0.8576,
"grad_norm": 6.973728656768799,
"kl": 0.34130859375,
"learning_rate": 1e-06,
"loss": 0.012,
"num_tokens": 14115125.0,
"reward": 8.21722412109375,
"reward_std": 5.218785285949707,
"rewards/rm_reward_func/mean": 8.21722412109375,
"rewards/rm_reward_func/std": 12.76237964630127,
"step": 1072
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 368.53125,
"completions/mean_terminated_length": 312.39129638671875,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.8584,
"grad_norm": 6.441647052764893,
"kl": 0.60107421875,
"learning_rate": 1e-06,
"loss": 0.0165,
"num_tokens": 14129646.0,
"reward": 8.17431640625,
"reward_std": 4.927266597747803,
"rewards/rm_reward_func/mean": 8.17431640625,
"rewards/rm_reward_func/std": 10.222173690795898,
"step": 1073
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 511.0,
"completions/mean_length": 328.34375,
"completions/mean_terminated_length": 267.125,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.8592,
"grad_norm": 5.157811164855957,
"kl": 1.52880859375,
"learning_rate": 1e-06,
"loss": 0.0403,
"num_tokens": 14142593.0,
"reward": 3.81683349609375,
"reward_std": 3.48171329498291,
"rewards/rm_reward_func/mean": 3.81683349609375,
"rewards/rm_reward_func/std": 9.174951553344727,
"step": 1074
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 504.0,
"completions/mean_length": 347.3125,
"completions/mean_terminated_length": 272.4545593261719,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.86,
"grad_norm": 5.949108600616455,
"kl": 0.45263671875,
"learning_rate": 1e-06,
"loss": 0.0243,
"num_tokens": 14155811.0,
"reward": 14.94580078125,
"reward_std": 4.586613655090332,
"rewards/rm_reward_func/mean": 14.94580078125,
"rewards/rm_reward_func/std": 6.567379951477051,
"step": 1075
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 350.25,
"completions/mean_terminated_length": 312.923095703125,
"completions/min_length": 152.0,
"completions/min_terminated_length": 152.0,
"epoch": 0.8608,
"grad_norm": 7.930327415466309,
"kl": 0.419921875,
"learning_rate": 1e-06,
"loss": -0.0041,
"num_tokens": 14168835.0,
"reward": 2.20947265625,
"reward_std": 6.557795524597168,
"rewards/rm_reward_func/mean": 2.20947265625,
"rewards/rm_reward_func/std": 8.4446382522583,
"step": 1076
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 507.0,
"completions/mean_length": 386.21875,
"completions/mean_terminated_length": 344.29168701171875,
"completions/min_length": 180.0,
"completions/min_terminated_length": 180.0,
"epoch": 0.8616,
"grad_norm": 8.628700256347656,
"kl": 0.5888671875,
"learning_rate": 1e-06,
"loss": -0.0536,
"num_tokens": 14183226.0,
"reward": 10.73797607421875,
"reward_std": 5.025568008422852,
"rewards/rm_reward_func/mean": 10.73797607421875,
"rewards/rm_reward_func/std": 8.780284881591797,
"step": 1077
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 453.21875,
"completions/mean_terminated_length": 426.5,
"completions/min_length": 354.0,
"completions/min_terminated_length": 354.0,
"epoch": 0.8624,
"grad_norm": 6.511486053466797,
"kl": 0.40283203125,
"learning_rate": 1e-06,
"loss": 0.033,
"num_tokens": 14200257.0,
"reward": 15.1558837890625,
"reward_std": 4.903788089752197,
"rewards/rm_reward_func/mean": 15.1558837890625,
"rewards/rm_reward_func/std": 12.219951629638672,
"step": 1078
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 492.0,
"completions/mean_length": 311.78125,
"completions/mean_terminated_length": 305.32257080078125,
"completions/min_length": 177.0,
"completions/min_terminated_length": 177.0,
"epoch": 0.8632,
"grad_norm": 9.137861251831055,
"kl": 1.89208984375,
"learning_rate": 1e-06,
"loss": 0.0506,
"num_tokens": 14212794.0,
"reward": 3.208251953125,
"reward_std": 6.694586753845215,
"rewards/rm_reward_func/mean": 3.208251953125,
"rewards/rm_reward_func/std": 14.088113784790039,
"step": 1079
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 492.0,
"completions/mean_length": 248.1875,
"completions/mean_terminated_length": 230.60000610351562,
"completions/min_length": 64.0,
"completions/min_terminated_length": 64.0,
"epoch": 0.864,
"grad_norm": 8.597902297973633,
"kl": 3.4228515625,
"learning_rate": 1e-06,
"loss": 0.1484,
"num_tokens": 14223552.0,
"reward": -1.8214111328125,
"reward_std": 4.2023773193359375,
"rewards/rm_reward_func/mean": -1.8214111328125,
"rewards/rm_reward_func/std": 9.527159690856934,
"step": 1080
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.40625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 369.84375,
"completions/mean_terminated_length": 272.5789489746094,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.8648,
"grad_norm": 5.029150485992432,
"kl": 0.5234375,
"learning_rate": 1e-06,
"loss": 0.0322,
"num_tokens": 14238843.0,
"reward": -7.0312652587890625,
"reward_std": 2.070448637008667,
"rewards/rm_reward_func/mean": -7.0312652587890625,
"rewards/rm_reward_func/std": 11.313837051391602,
"step": 1081
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 444.0,
"completions/max_terminated_length": 444.0,
"completions/mean_length": 282.59375,
"completions/mean_terminated_length": 282.59375,
"completions/min_length": 79.0,
"completions/min_terminated_length": 79.0,
"epoch": 0.8656,
"grad_norm": 16.06825828552246,
"kl": 5.65869140625,
"learning_rate": 1e-06,
"loss": 0.2783,
"num_tokens": 14251942.0,
"reward": 3.9051666259765625,
"reward_std": 4.887936592102051,
"rewards/rm_reward_func/mean": 3.9051666259765625,
"rewards/rm_reward_func/std": 14.745894432067871,
"step": 1082
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 458.0,
"completions/mean_length": 306.90625,
"completions/mean_terminated_length": 300.2903137207031,
"completions/min_length": 36.0,
"completions/min_terminated_length": 36.0,
"epoch": 0.8664,
"grad_norm": 7.322081089019775,
"kl": 3.85546875,
"learning_rate": 1e-06,
"loss": 0.1572,
"num_tokens": 14267451.0,
"reward": 8.35272216796875,
"reward_std": 9.634117126464844,
"rewards/rm_reward_func/mean": 8.35272216796875,
"rewards/rm_reward_func/std": 19.10032844543457,
"step": 1083
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 345.5,
"completions/mean_terminated_length": 340.1290283203125,
"completions/min_length": 57.0,
"completions/min_terminated_length": 57.0,
"epoch": 0.8672,
"grad_norm": 12.525562286376953,
"kl": 5.53955078125,
"learning_rate": 1e-06,
"loss": 0.4021,
"num_tokens": 14285251.0,
"reward": 14.5098876953125,
"reward_std": 6.451245307922363,
"rewards/rm_reward_func/mean": 14.5098876953125,
"rewards/rm_reward_func/std": 12.280783653259277,
"step": 1084
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 491.0,
"completions/mean_length": 227.1875,
"completions/mean_terminated_length": 197.72413635253906,
"completions/min_length": 21.0,
"completions/min_terminated_length": 21.0,
"epoch": 0.868,
"grad_norm": 12.410992622375488,
"kl": 6.01171875,
"learning_rate": 1e-06,
"loss": 0.2138,
"num_tokens": 14295113.0,
"reward": 2.096435546875,
"reward_std": 6.907596588134766,
"rewards/rm_reward_func/mean": 2.096435546875,
"rewards/rm_reward_func/std": 11.79339599609375,
"step": 1085
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 475.0,
"completions/mean_length": 263.28125,
"completions/mean_terminated_length": 246.70001220703125,
"completions/min_length": 58.0,
"completions/min_terminated_length": 58.0,
"epoch": 0.8688,
"grad_norm": 7.831367015838623,
"kl": 0.89599609375,
"learning_rate": 1e-06,
"loss": -0.026,
"num_tokens": 14308474.0,
"reward": 2.3360595703125,
"reward_std": 6.861347675323486,
"rewards/rm_reward_func/mean": 2.3360595703125,
"rewards/rm_reward_func/std": 9.401144027709961,
"step": 1086
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 496.0,
"completions/mean_length": 246.40625,
"completions/mean_terminated_length": 237.8386993408203,
"completions/min_length": 32.0,
"completions/min_terminated_length": 32.0,
"epoch": 0.8696,
"grad_norm": 12.294774055480957,
"kl": 3.60009765625,
"learning_rate": 1e-06,
"loss": 0.0805,
"num_tokens": 14319327.0,
"reward": 1.0987548828125,
"reward_std": 5.113190650939941,
"rewards/rm_reward_func/mean": 1.0987548828125,
"rewards/rm_reward_func/std": 7.818641185760498,
"step": 1087
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 479.0,
"completions/mean_length": 269.0,
"completions/mean_terminated_length": 261.1612854003906,
"completions/min_length": 68.0,
"completions/min_terminated_length": 68.0,
"epoch": 0.8704,
"grad_norm": 13.56505298614502,
"kl": 4.18359375,
"learning_rate": 1e-06,
"loss": 0.114,
"num_tokens": 14331231.0,
"reward": 1.3173828125,
"reward_std": 7.998728275299072,
"rewards/rm_reward_func/mean": 1.3173828125,
"rewards/rm_reward_func/std": 18.137277603149414,
"step": 1088
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 472.0,
"completions/mean_length": 299.0625,
"completions/mean_terminated_length": 249.92308044433594,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.8712,
"grad_norm": 5.182938575744629,
"kl": 1.154296875,
"learning_rate": 1e-06,
"loss": 0.0724,
"num_tokens": 14343945.0,
"reward": 9.21612548828125,
"reward_std": 8.317728042602539,
"rewards/rm_reward_func/mean": 9.21612548828125,
"rewards/rm_reward_func/std": 16.426794052124023,
"step": 1089
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 403.0,
"completions/max_terminated_length": 403.0,
"completions/mean_length": 185.0,
"completions/mean_terminated_length": 185.0,
"completions/min_length": 37.0,
"completions/min_terminated_length": 37.0,
"epoch": 0.872,
"grad_norm": 68.98828125,
"kl": 4.005859375,
"learning_rate": 1e-06,
"loss": 0.2527,
"num_tokens": 14354649.0,
"reward": -9.63653564453125,
"reward_std": 3.627918243408203,
"rewards/rm_reward_func/mean": -9.63653564453125,
"rewards/rm_reward_func/std": 8.9398775100708,
"step": 1090
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 488.0,
"completions/max_terminated_length": 488.0,
"completions/mean_length": 313.59375,
"completions/mean_terminated_length": 313.59375,
"completions/min_length": 111.0,
"completions/min_terminated_length": 111.0,
"epoch": 0.8728,
"grad_norm": 13.77872085571289,
"kl": 5.103515625,
"learning_rate": 1e-06,
"loss": 0.2633,
"num_tokens": 14367212.0,
"reward": 0.75341796875,
"reward_std": 5.394412040710449,
"rewards/rm_reward_func/mean": 0.75341796875,
"rewards/rm_reward_func/std": 13.423392295837402,
"step": 1091
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 239.0,
"completions/mean_length": 228.75,
"completions/mean_terminated_length": 134.33334350585938,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.8736,
"grad_norm": 11.227128028869629,
"kl": 1.6025390625,
"learning_rate": 1e-06,
"loss": 0.1076,
"num_tokens": 14378772.0,
"reward": -0.025146484375,
"reward_std": 2.094247341156006,
"rewards/rm_reward_func/mean": -0.025146484375,
"rewards/rm_reward_func/std": 8.626964569091797,
"step": 1092
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 347.5,
"completions/mean_terminated_length": 283.13043212890625,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.8744,
"grad_norm": 6.2191619873046875,
"kl": 3.33203125,
"learning_rate": 1e-06,
"loss": 0.1357,
"num_tokens": 14394876.0,
"reward": 16.056884765625,
"reward_std": 9.725163459777832,
"rewards/rm_reward_func/mean": 16.056884765625,
"rewards/rm_reward_func/std": 21.10055923461914,
"step": 1093
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 407.0,
"completions/max_terminated_length": 407.0,
"completions/mean_length": 238.3125,
"completions/mean_terminated_length": 238.3125,
"completions/min_length": 75.0,
"completions/min_terminated_length": 75.0,
"epoch": 0.8752,
"grad_norm": 9.5230073928833,
"kl": 3.8193359375,
"learning_rate": 1e-06,
"loss": 0.0945,
"num_tokens": 14404414.0,
"reward": -4.79345703125,
"reward_std": 5.990765571594238,
"rewards/rm_reward_func/mean": -4.79345703125,
"rewards/rm_reward_func/std": 13.47055721282959,
"step": 1094
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 505.0,
"completions/mean_length": 236.90625,
"completions/mean_terminated_length": 185.9629669189453,
"completions/min_length": 74.0,
"completions/min_terminated_length": 74.0,
"epoch": 0.876,
"grad_norm": 7.131078243255615,
"kl": 1.12109375,
"learning_rate": 1e-06,
"loss": 0.0133,
"num_tokens": 14415475.0,
"reward": 0.505615234375,
"reward_std": 3.531787395477295,
"rewards/rm_reward_func/mean": 0.505615234375,
"rewards/rm_reward_func/std": 6.195098876953125,
"step": 1095
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 386.0,
"completions/max_terminated_length": 386.0,
"completions/mean_length": 217.75,
"completions/mean_terminated_length": 217.75,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.8768,
"grad_norm": 7.909564971923828,
"kl": 1.53125,
"learning_rate": 1e-06,
"loss": 0.0247,
"num_tokens": 14425171.0,
"reward": -1.14208984375,
"reward_std": 1.705171823501587,
"rewards/rm_reward_func/mean": -1.14208984375,
"rewards/rm_reward_func/std": 10.815631866455078,
"step": 1096
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 265.15625,
"completions/mean_terminated_length": 182.875,
"completions/min_length": 73.0,
"completions/min_terminated_length": 73.0,
"epoch": 0.8776,
"grad_norm": 4.670645236968994,
"kl": 2.76171875,
"learning_rate": 1e-06,
"loss": 0.1654,
"num_tokens": 14438568.0,
"reward": 10.848876953125,
"reward_std": 4.882187366485596,
"rewards/rm_reward_func/mean": 10.848876953125,
"rewards/rm_reward_func/std": 13.137846946716309,
"step": 1097
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 390.0,
"completions/mean_length": 317.75,
"completions/mean_terminated_length": 253.0,
"completions/min_length": 124.0,
"completions/min_terminated_length": 124.0,
"epoch": 0.8784,
"grad_norm": 11.463107109069824,
"kl": 3.96484375,
"learning_rate": 1e-06,
"loss": 0.1702,
"num_tokens": 14451368.0,
"reward": 4.6419677734375,
"reward_std": 3.999204397201538,
"rewards/rm_reward_func/mean": 4.6419677734375,
"rewards/rm_reward_func/std": 16.210906982421875,
"step": 1098
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 488.0,
"completions/max_terminated_length": 488.0,
"completions/mean_length": 302.15625,
"completions/mean_terminated_length": 302.15625,
"completions/min_length": 56.0,
"completions/min_terminated_length": 56.0,
"epoch": 0.8792,
"grad_norm": 21.74379539489746,
"kl": 7.875,
"learning_rate": 1e-06,
"loss": 0.225,
"num_tokens": 14463477.0,
"reward": -5.63720703125,
"reward_std": 9.232072830200195,
"rewards/rm_reward_func/mean": -5.63720703125,
"rewards/rm_reward_func/std": 18.789886474609375,
"step": 1099
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 299.25,
"completions/mean_terminated_length": 259.85186767578125,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.88,
"grad_norm": 6.47990608215332,
"kl": 5.8642578125,
"learning_rate": 1e-06,
"loss": 0.1774,
"num_tokens": 14476309.0,
"reward": -0.43701171875,
"reward_std": 10.200658798217773,
"rewards/rm_reward_func/mean": -0.43701171875,
"rewards/rm_reward_func/std": 14.860036849975586,
"step": 1100
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 499.0,
"completions/max_terminated_length": 499.0,
"completions/mean_length": 105.0,
"completions/mean_terminated_length": 105.0,
"completions/min_length": 63.0,
"completions/min_terminated_length": 63.0,
"epoch": 0.8808,
"grad_norm": 5.006459712982178,
"kl": 1.625,
"learning_rate": 1e-06,
"loss": -0.0308,
"num_tokens": 14483213.0,
"reward": 1.8359375,
"reward_std": 0.8900178074836731,
"rewards/rm_reward_func/mean": 1.8359375,
"rewards/rm_reward_func/std": 11.928114891052246,
"step": 1101
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 280.28125,
"completions/mean_terminated_length": 264.8333435058594,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.8816,
"grad_norm": 8.10487174987793,
"kl": 2.853515625,
"learning_rate": 1e-06,
"loss": 0.0011,
"num_tokens": 14495190.0,
"reward": 5.513671875,
"reward_std": 8.272500991821289,
"rewards/rm_reward_func/mean": 5.513671875,
"rewards/rm_reward_func/std": 16.234622955322266,
"step": 1102
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 507.0,
"completions/mean_length": 239.5,
"completions/mean_terminated_length": 230.7096710205078,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.8824,
"grad_norm": 3.8931567668914795,
"kl": 0.5224609375,
"learning_rate": 1e-06,
"loss": 0.0255,
"num_tokens": 14505822.0,
"reward": 11.3662109375,
"reward_std": 2.4082765579223633,
"rewards/rm_reward_func/mean": 11.3662109375,
"rewards/rm_reward_func/std": 5.015256404876709,
"step": 1103
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 492.0,
"completions/mean_length": 237.28125,
"completions/mean_terminated_length": 198.0357208251953,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.8832,
"grad_norm": 10.963964462280273,
"kl": 4.1640625,
"learning_rate": 1e-06,
"loss": 0.1059,
"num_tokens": 14518447.0,
"reward": 1.65234375,
"reward_std": 3.4185004234313965,
"rewards/rm_reward_func/mean": 1.65234375,
"rewards/rm_reward_func/std": 13.96972370147705,
"step": 1104
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 469.0,
"completions/max_terminated_length": 469.0,
"completions/mean_length": 252.53125,
"completions/mean_terminated_length": 252.53125,
"completions/min_length": 83.0,
"completions/min_terminated_length": 83.0,
"epoch": 0.884,
"grad_norm": 7.281085014343262,
"kl": 4.080078125,
"learning_rate": 1e-06,
"loss": 0.212,
"num_tokens": 14529264.0,
"reward": -5.43353271484375,
"reward_std": 5.849520206451416,
"rewards/rm_reward_func/mean": -5.43353271484375,
"rewards/rm_reward_func/std": 9.325867652893066,
"step": 1105
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 492.0,
"completions/max_terminated_length": 492.0,
"completions/mean_length": 341.4375,
"completions/mean_terminated_length": 341.4375,
"completions/min_length": 202.0,
"completions/min_terminated_length": 202.0,
"epoch": 0.8848,
"grad_norm": 7.2115478515625,
"kl": 0.4150390625,
"learning_rate": 1e-06,
"loss": 0.0024,
"num_tokens": 14543294.0,
"reward": 7.579345703125,
"reward_std": 4.300886154174805,
"rewards/rm_reward_func/mean": 7.579345703125,
"rewards/rm_reward_func/std": 5.634270191192627,
"step": 1106
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 450.0,
"completions/max_terminated_length": 450.0,
"completions/mean_length": 281.21875,
"completions/mean_terminated_length": 281.21875,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.8856,
"grad_norm": 11.654593467712402,
"kl": 1.85693359375,
"learning_rate": 1e-06,
"loss": 0.0569,
"num_tokens": 14554965.0,
"reward": 2.6845703125,
"reward_std": 5.363699436187744,
"rewards/rm_reward_func/mean": 2.6845703125,
"rewards/rm_reward_func/std": 11.830452919006348,
"step": 1107
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 509.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 215.78125,
"completions/mean_terminated_length": 215.78125,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.8864,
"grad_norm": 7.386474132537842,
"kl": 0.47412109375,
"learning_rate": 1e-06,
"loss": -0.0309,
"num_tokens": 14564774.0,
"reward": 16.75201416015625,
"reward_std": 2.713106155395508,
"rewards/rm_reward_func/mean": 16.75201416015625,
"rewards/rm_reward_func/std": 14.134072303771973,
"step": 1108
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 505.0,
"completions/mean_length": 278.9375,
"completions/mean_terminated_length": 245.6428680419922,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.8872,
"grad_norm": 8.896984100341797,
"kl": 0.71337890625,
"learning_rate": 1e-06,
"loss": -0.012,
"num_tokens": 14576156.0,
"reward": 3.846038818359375,
"reward_std": 4.625146389007568,
"rewards/rm_reward_func/mean": 3.846038818359375,
"rewards/rm_reward_func/std": 7.2643961906433105,
"step": 1109
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 401.84375,
"completions/mean_terminated_length": 386.1071472167969,
"completions/min_length": 247.0,
"completions/min_terminated_length": 247.0,
"epoch": 0.888,
"grad_norm": 9.240114212036133,
"kl": 1.05859375,
"learning_rate": 1e-06,
"loss": 0.0088,
"num_tokens": 14591575.0,
"reward": 8.68145751953125,
"reward_std": 6.467904567718506,
"rewards/rm_reward_func/mean": 8.68145751953125,
"rewards/rm_reward_func/std": 9.515111923217773,
"step": 1110
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 397.84375,
"completions/mean_terminated_length": 345.9545593261719,
"completions/min_length": 235.0,
"completions/min_terminated_length": 235.0,
"epoch": 0.8888,
"grad_norm": 6.770890235900879,
"kl": 2.41015625,
"learning_rate": 1e-06,
"loss": 0.0554,
"num_tokens": 14606562.0,
"reward": -2.921142578125,
"reward_std": 5.699788570404053,
"rewards/rm_reward_func/mean": -2.921142578125,
"rewards/rm_reward_func/std": 17.231367111206055,
"step": 1111
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 480.0,
"completions/mean_length": 187.0,
"completions/mean_terminated_length": 96.0,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.8896,
"grad_norm": 4.088595390319824,
"kl": 0.5556640625,
"learning_rate": 1e-06,
"loss": 0.025,
"num_tokens": 14618274.0,
"reward": 4.689453125,
"reward_std": 0.7515532970428467,
"rewards/rm_reward_func/mean": 4.689453125,
"rewards/rm_reward_func/std": 11.583274841308594,
"step": 1112
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 500.0,
"completions/mean_length": 283.46875,
"completions/mean_terminated_length": 179.59091186523438,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.8904,
"grad_norm": 5.539703369140625,
"kl": 2.7138671875,
"learning_rate": 1e-06,
"loss": 0.1386,
"num_tokens": 14631761.0,
"reward": 2.4609222412109375,
"reward_std": 4.746338844299316,
"rewards/rm_reward_func/mean": 2.4609222412109375,
"rewards/rm_reward_func/std": 9.370748519897461,
"step": 1113
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 264.59375,
"completions/mean_terminated_length": 248.10000610351562,
"completions/min_length": 75.0,
"completions/min_terminated_length": 75.0,
"epoch": 0.8912,
"grad_norm": 16.168067932128906,
"kl": 1.09033203125,
"learning_rate": 1e-06,
"loss": -0.1252,
"num_tokens": 14643388.0,
"reward": 1.837890625,
"reward_std": 7.579433441162109,
"rewards/rm_reward_func/mean": 1.837890625,
"rewards/rm_reward_func/std": 13.530681610107422,
"step": 1114
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 501.0,
"completions/mean_length": 415.03125,
"completions/mean_terminated_length": 364.23809814453125,
"completions/min_length": 60.0,
"completions/min_terminated_length": 60.0,
"epoch": 0.892,
"grad_norm": 15.032100677490234,
"kl": 4.005859375,
"learning_rate": 1e-06,
"loss": 0.101,
"num_tokens": 14662877.0,
"reward": -12.3818359375,
"reward_std": 5.061200141906738,
"rewards/rm_reward_func/mean": -12.3818359375,
"rewards/rm_reward_func/std": 8.946840286254883,
"step": 1115
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 503.0,
"completions/mean_length": 241.1875,
"completions/mean_terminated_length": 213.1724090576172,
"completions/min_length": 33.0,
"completions/min_terminated_length": 33.0,
"epoch": 0.8928,
"grad_norm": 12.395576477050781,
"kl": 3.50927734375,
"learning_rate": 1e-06,
"loss": 0.3018,
"num_tokens": 14673483.0,
"reward": -4.934326171875,
"reward_std": 6.060213088989258,
"rewards/rm_reward_func/mean": -4.934326171875,
"rewards/rm_reward_func/std": 12.644729614257812,
"step": 1116
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 503.0,
"completions/mean_length": 316.15625,
"completions/mean_terminated_length": 250.875,
"completions/min_length": 107.0,
"completions/min_terminated_length": 107.0,
"epoch": 0.8936,
"grad_norm": 9.258816719055176,
"kl": 4.6328125,
"learning_rate": 1e-06,
"loss": 0.1356,
"num_tokens": 14685808.0,
"reward": -9.6767578125,
"reward_std": 4.323359489440918,
"rewards/rm_reward_func/mean": -9.6767578125,
"rewards/rm_reward_func/std": 4.797229290008545,
"step": 1117
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 347.75,
"completions/mean_terminated_length": 336.8000183105469,
"completions/min_length": 150.0,
"completions/min_terminated_length": 150.0,
"epoch": 0.8944,
"grad_norm": 13.36823558807373,
"kl": 3.09423828125,
"learning_rate": 1e-06,
"loss": 0.0864,
"num_tokens": 14698928.0,
"reward": 3.083251953125,
"reward_std": 6.492587089538574,
"rewards/rm_reward_func/mean": 3.083251953125,
"rewards/rm_reward_func/std": 15.465557098388672,
"step": 1118
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 497.0,
"completions/mean_length": 344.90625,
"completions/mean_terminated_length": 327.6206970214844,
"completions/min_length": 94.0,
"completions/min_terminated_length": 94.0,
"epoch": 0.8952,
"grad_norm": 16.696882247924805,
"kl": 1.796875,
"learning_rate": 1e-06,
"loss": 0.026,
"num_tokens": 14714893.0,
"reward": 4.2606201171875,
"reward_std": 6.969727516174316,
"rewards/rm_reward_func/mean": 4.2606201171875,
"rewards/rm_reward_func/std": 21.33582305908203,
"step": 1119
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 372.0,
"completions/mean_length": 211.4375,
"completions/mean_terminated_length": 201.74192810058594,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.896,
"grad_norm": 36.645877838134766,
"kl": 3.154296875,
"learning_rate": 1e-06,
"loss": 0.3064,
"num_tokens": 14726907.0,
"reward": -5.021240234375,
"reward_std": 5.98872184753418,
"rewards/rm_reward_func/mean": -5.021240234375,
"rewards/rm_reward_func/std": 11.189291000366211,
"step": 1120
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 410.0,
"completions/max_terminated_length": 410.0,
"completions/mean_length": 206.65625,
"completions/mean_terminated_length": 206.65625,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.8968,
"grad_norm": 5.673826217651367,
"kl": 0.5107421875,
"learning_rate": 1e-06,
"loss": -0.0006,
"num_tokens": 14739032.0,
"reward": 10.79296875,
"reward_std": 1.6997040510177612,
"rewards/rm_reward_func/mean": 10.79296875,
"rewards/rm_reward_func/std": 6.225403785705566,
"step": 1121
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 308.21875,
"completions/mean_terminated_length": 270.4814758300781,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.8976,
"grad_norm": 9.43205451965332,
"kl": 4.9072265625,
"learning_rate": 1e-06,
"loss": 0.214,
"num_tokens": 14751527.0,
"reward": -3.093505859375,
"reward_std": 5.163877487182617,
"rewards/rm_reward_func/mean": -3.093505859375,
"rewards/rm_reward_func/std": 11.953996658325195,
"step": 1122
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 342.0,
"completions/mean_length": 182.8125,
"completions/mean_terminated_length": 135.7857208251953,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.8984,
"grad_norm": 12.99282169342041,
"kl": 2.005859375,
"learning_rate": 1e-06,
"loss": 0.2826,
"num_tokens": 14761121.0,
"reward": 0.334228515625,
"reward_std": 4.173687934875488,
"rewards/rm_reward_func/mean": 0.334228515625,
"rewards/rm_reward_func/std": 6.381908416748047,
"step": 1123
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 185.0625,
"completions/mean_terminated_length": 124.51851654052734,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.8992,
"grad_norm": 9.33836841583252,
"kl": 0.52001953125,
"learning_rate": 1e-06,
"loss": 0.022,
"num_tokens": 14770411.0,
"reward": 7.060302734375,
"reward_std": 1.1086238622665405,
"rewards/rm_reward_func/mean": 7.060302734375,
"rewards/rm_reward_func/std": 5.903942584991455,
"step": 1124
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 345.09375,
"completions/mean_terminated_length": 321.25,
"completions/min_length": 166.0,
"completions/min_terminated_length": 166.0,
"epoch": 0.9,
"grad_norm": 10.884320259094238,
"kl": 2.587890625,
"learning_rate": 1e-06,
"loss": 0.0485,
"num_tokens": 14786686.0,
"reward": 11.99560546875,
"reward_std": 7.212575912475586,
"rewards/rm_reward_func/mean": 11.99560546875,
"rewards/rm_reward_func/std": 22.268768310546875,
"step": 1125
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 448.0,
"completions/max_terminated_length": 448.0,
"completions/mean_length": 232.03125,
"completions/mean_terminated_length": 232.03125,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9008,
"grad_norm": 9.929396629333496,
"kl": 1.83203125,
"learning_rate": 1e-06,
"loss": 0.0425,
"num_tokens": 14795951.0,
"reward": -1.327392578125,
"reward_std": 3.5014753341674805,
"rewards/rm_reward_func/mean": -1.327392578125,
"rewards/rm_reward_func/std": 10.548651695251465,
"step": 1126
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 410.625,
"completions/mean_terminated_length": 357.5238037109375,
"completions/min_length": 252.0,
"completions/min_terminated_length": 252.0,
"epoch": 0.9016,
"grad_norm": 10.775675773620605,
"kl": 1.833740234375,
"learning_rate": 1e-06,
"loss": 0.0623,
"num_tokens": 14812659.0,
"reward": 6.20263671875,
"reward_std": 6.819557189941406,
"rewards/rm_reward_func/mean": 6.20263671875,
"rewards/rm_reward_func/std": 21.737092971801758,
"step": 1127
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.46875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 476.0,
"completions/mean_length": 370.125,
"completions/mean_terminated_length": 244.94117736816406,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9024,
"grad_norm": 8.568568229675293,
"kl": 1.802734375,
"learning_rate": 1e-06,
"loss": 0.0395,
"num_tokens": 14826751.0,
"reward": -0.14697265625,
"reward_std": 5.660521507263184,
"rewards/rm_reward_func/mean": -0.14697265625,
"rewards/rm_reward_func/std": 9.781434059143066,
"step": 1128
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 494.0,
"completions/mean_length": 389.375,
"completions/mean_terminated_length": 333.6363830566406,
"completions/min_length": 131.0,
"completions/min_terminated_length": 131.0,
"epoch": 0.9032,
"grad_norm": 14.49055004119873,
"kl": 1.3701171875,
"learning_rate": 1e-06,
"loss": 0.1125,
"num_tokens": 14842283.0,
"reward": -4.70123291015625,
"reward_std": 5.61661434173584,
"rewards/rm_reward_func/mean": -4.70123291015625,
"rewards/rm_reward_func/std": 6.281442165374756,
"step": 1129
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 377.125,
"completions/mean_terminated_length": 357.8571472167969,
"completions/min_length": 132.0,
"completions/min_terminated_length": 132.0,
"epoch": 0.904,
"grad_norm": 8.899121284484863,
"kl": 0.798828125,
"learning_rate": 1e-06,
"loss": -0.0292,
"num_tokens": 14856479.0,
"reward": 11.397480010986328,
"reward_std": 5.290931701660156,
"rewards/rm_reward_func/mean": 11.397480010986328,
"rewards/rm_reward_func/std": 16.51535415649414,
"step": 1130
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 503.0,
"completions/mean_length": 404.15625,
"completions/mean_terminated_length": 361.95654296875,
"completions/min_length": 45.0,
"completions/min_terminated_length": 45.0,
"epoch": 0.9048,
"grad_norm": 20.565969467163086,
"kl": 2.8994140625,
"learning_rate": 1e-06,
"loss": -0.0021,
"num_tokens": 14872308.0,
"reward": -4.064208984375,
"reward_std": 9.170465469360352,
"rewards/rm_reward_func/mean": -4.064208984375,
"rewards/rm_reward_func/std": 11.234378814697266,
"step": 1131
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 487.0,
"completions/mean_length": 282.34375,
"completions/mean_terminated_length": 267.0333557128906,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9056,
"grad_norm": 38.7857551574707,
"kl": 1.5107421875,
"learning_rate": 1e-06,
"loss": 0.0238,
"num_tokens": 14884023.0,
"reward": 4.152587890625,
"reward_std": 8.475656509399414,
"rewards/rm_reward_func/mean": 4.152587890625,
"rewards/rm_reward_func/std": 13.332249641418457,
"step": 1132
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 248.625,
"completions/mean_terminated_length": 221.37930297851562,
"completions/min_length": 37.0,
"completions/min_terminated_length": 37.0,
"epoch": 0.9064,
"grad_norm": 13.033285140991211,
"kl": 1.998046875,
"learning_rate": 1e-06,
"loss": 0.0814,
"num_tokens": 14894971.0,
"reward": 2.286590576171875,
"reward_std": 8.568852424621582,
"rewards/rm_reward_func/mean": 2.286590576171875,
"rewards/rm_reward_func/std": 21.08578109741211,
"step": 1133
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 493.0,
"completions/mean_length": 348.375,
"completions/mean_terminated_length": 343.0967712402344,
"completions/min_length": 141.0,
"completions/min_terminated_length": 141.0,
"epoch": 0.9072,
"grad_norm": 13.447713851928711,
"kl": 5.14453125,
"learning_rate": 1e-06,
"loss": 0.2624,
"num_tokens": 14908463.0,
"reward": -6.699335098266602,
"reward_std": 6.876185417175293,
"rewards/rm_reward_func/mean": -6.699335098266602,
"rewards/rm_reward_func/std": 8.872479438781738,
"step": 1134
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 502.0,
"completions/mean_length": 366.78125,
"completions/mean_terminated_length": 339.8888854980469,
"completions/min_length": 74.0,
"completions/min_terminated_length": 74.0,
"epoch": 0.908,
"grad_norm": 28.221643447875977,
"kl": 3.7958984375,
"learning_rate": 1e-06,
"loss": 0.0659,
"num_tokens": 14922760.0,
"reward": -6.485260009765625,
"reward_std": 7.134948253631592,
"rewards/rm_reward_func/mean": -6.485260009765625,
"rewards/rm_reward_func/std": 11.682751655578613,
"step": 1135
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 482.0,
"completions/mean_length": 265.59375,
"completions/mean_terminated_length": 208.73077392578125,
"completions/min_length": 66.0,
"completions/min_terminated_length": 66.0,
"epoch": 0.9088,
"grad_norm": 101.20088195800781,
"kl": 2.15234375,
"learning_rate": 1e-06,
"loss": -0.1044,
"num_tokens": 14937771.0,
"reward": 0.34814453125,
"reward_std": 8.503646850585938,
"rewards/rm_reward_func/mean": 0.34814453125,
"rewards/rm_reward_func/std": 12.46761703491211,
"step": 1136
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 467.0,
"completions/mean_length": 288.5625,
"completions/mean_terminated_length": 273.66668701171875,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9096,
"grad_norm": 10.237065315246582,
"kl": 1.3642578125,
"learning_rate": 1e-06,
"loss": 0.0369,
"num_tokens": 14951197.0,
"reward": 0.99951171875,
"reward_std": 5.672191143035889,
"rewards/rm_reward_func/mean": 0.99951171875,
"rewards/rm_reward_func/std": 10.137231826782227,
"step": 1137
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 492.0,
"completions/mean_length": 261.21875,
"completions/mean_terminated_length": 225.3928680419922,
"completions/min_length": 37.0,
"completions/min_terminated_length": 37.0,
"epoch": 0.9104,
"grad_norm": 19.81197166442871,
"kl": 4.25390625,
"learning_rate": 1e-06,
"loss": 0.088,
"num_tokens": 14961988.0,
"reward": -4.6865234375,
"reward_std": 8.24875259399414,
"rewards/rm_reward_func/mean": -4.6865234375,
"rewards/rm_reward_func/std": 21.321937561035156,
"step": 1138
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 308.875,
"completions/mean_terminated_length": 287.862060546875,
"completions/min_length": 53.0,
"completions/min_terminated_length": 53.0,
"epoch": 0.9112,
"grad_norm": 17.55968475341797,
"kl": 4.6123046875,
"learning_rate": 1e-06,
"loss": 0.1737,
"num_tokens": 14974224.0,
"reward": -4.378509521484375,
"reward_std": 5.992432594299316,
"rewards/rm_reward_func/mean": -4.378509521484375,
"rewards/rm_reward_func/std": 11.200332641601562,
"step": 1139
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.53125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 507.0,
"completions/mean_length": 454.5,
"completions/mean_terminated_length": 389.3333435058594,
"completions/min_length": 162.0,
"completions/min_terminated_length": 162.0,
"epoch": 0.912,
"grad_norm": 86.18330383300781,
"kl": 3.326171875,
"learning_rate": 1e-06,
"loss": 0.0577,
"num_tokens": 14992072.0,
"reward": -0.056640625,
"reward_std": 9.960199356079102,
"rewards/rm_reward_func/mean": -0.056640625,
"rewards/rm_reward_func/std": 13.414643287658691,
"step": 1140
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 504.0,
"completions/max_terminated_length": 504.0,
"completions/mean_length": 271.09375,
"completions/mean_terminated_length": 271.09375,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9128,
"grad_norm": 25.556337356567383,
"kl": 5.330078125,
"learning_rate": 1e-06,
"loss": 0.2298,
"num_tokens": 15002867.0,
"reward": -7.34228515625,
"reward_std": 4.103965759277344,
"rewards/rm_reward_func/mean": -7.34228515625,
"rewards/rm_reward_func/std": 8.186371803283691,
"step": 1141
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 479.0,
"completions/max_terminated_length": 479.0,
"completions/mean_length": 199.3125,
"completions/mean_terminated_length": 199.3125,
"completions/min_length": 21.0,
"completions/min_terminated_length": 21.0,
"epoch": 0.9136,
"grad_norm": 33.93816375732422,
"kl": 5.087890625,
"learning_rate": 1e-06,
"loss": 0.1551,
"num_tokens": 15011485.0,
"reward": -7.401123046875,
"reward_std": 7.4005818367004395,
"rewards/rm_reward_func/mean": -7.401123046875,
"rewards/rm_reward_func/std": 17.179683685302734,
"step": 1142
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 463.0,
"completions/mean_length": 250.84375,
"completions/mean_terminated_length": 190.57693481445312,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9144,
"grad_norm": 22.07563591003418,
"kl": 4.1875,
"learning_rate": 1e-06,
"loss": 0.069,
"num_tokens": 15025976.0,
"reward": 0.76611328125,
"reward_std": 4.296019077301025,
"rewards/rm_reward_func/mean": 0.76611328125,
"rewards/rm_reward_func/std": 12.173991203308105,
"step": 1143
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 435.0,
"completions/max_terminated_length": 435.0,
"completions/mean_length": 278.59375,
"completions/mean_terminated_length": 278.59375,
"completions/min_length": 89.0,
"completions/min_terminated_length": 89.0,
"epoch": 0.9152,
"grad_norm": 22.454143524169922,
"kl": 2.92822265625,
"learning_rate": 1e-06,
"loss": -0.0074,
"num_tokens": 15041539.0,
"reward": -3.9213485717773438,
"reward_std": 7.493886470794678,
"rewards/rm_reward_func/mean": -3.9213485717773438,
"rewards/rm_reward_func/std": 12.379457473754883,
"step": 1144
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 511.0,
"completions/mean_length": 385.40625,
"completions/mean_terminated_length": 343.2083435058594,
"completions/min_length": 59.0,
"completions/min_terminated_length": 59.0,
"epoch": 0.916,
"grad_norm": 27.009363174438477,
"kl": 4.2822265625,
"learning_rate": 1e-06,
"loss": 0.1094,
"num_tokens": 15056096.0,
"reward": -11.80712890625,
"reward_std": 8.341408729553223,
"rewards/rm_reward_func/mean": -11.80712890625,
"rewards/rm_reward_func/std": 11.495277404785156,
"step": 1145
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 479.0,
"completions/mean_length": 266.65625,
"completions/mean_terminated_length": 258.7419128417969,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9168,
"grad_norm": 43.986976623535156,
"kl": 1.9150390625,
"learning_rate": 1e-06,
"loss": -0.0023,
"num_tokens": 15066469.0,
"reward": 4.2705078125,
"reward_std": 8.840106964111328,
"rewards/rm_reward_func/mean": 4.2705078125,
"rewards/rm_reward_func/std": 16.68376922607422,
"step": 1146
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 482.0,
"completions/mean_length": 333.5625,
"completions/mean_terminated_length": 300.5185241699219,
"completions/min_length": 41.0,
"completions/min_terminated_length": 41.0,
"epoch": 0.9176,
"grad_norm": 10.10371208190918,
"kl": 1.15283203125,
"learning_rate": 1e-06,
"loss": -0.0467,
"num_tokens": 15084015.0,
"reward": 10.65411376953125,
"reward_std": 9.344249725341797,
"rewards/rm_reward_func/mean": 10.65411376953125,
"rewards/rm_reward_func/std": 16.203916549682617,
"step": 1147
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 501.0,
"completions/mean_length": 311.8125,
"completions/mean_terminated_length": 274.7407531738281,
"completions/min_length": 82.0,
"completions/min_terminated_length": 82.0,
"epoch": 0.9184,
"grad_norm": 36.31003952026367,
"kl": 3.45703125,
"learning_rate": 1e-06,
"loss": 0.0783,
"num_tokens": 15096569.0,
"reward": -10.46044921875,
"reward_std": 8.902938842773438,
"rewards/rm_reward_func/mean": -10.46044921875,
"rewards/rm_reward_func/std": 14.822985649108887,
"step": 1148
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.40625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 497.0,
"completions/mean_length": 440.59375,
"completions/mean_terminated_length": 391.7368469238281,
"completions/min_length": 273.0,
"completions/min_terminated_length": 273.0,
"epoch": 0.9192,
"grad_norm": 14.747289657592773,
"kl": 1.51904296875,
"learning_rate": 1e-06,
"loss": 0.0563,
"num_tokens": 15113956.0,
"reward": 6.316192626953125,
"reward_std": 6.612483501434326,
"rewards/rm_reward_func/mean": 6.316192626953125,
"rewards/rm_reward_func/std": 10.599157333374023,
"step": 1149
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 488.0,
"completions/max_terminated_length": 488.0,
"completions/mean_length": 261.78125,
"completions/mean_terminated_length": 261.78125,
"completions/min_length": 51.0,
"completions/min_terminated_length": 51.0,
"epoch": 0.92,
"grad_norm": 21.25505828857422,
"kl": 3.0498046875,
"learning_rate": 1e-06,
"loss": 0.0598,
"num_tokens": 15124293.0,
"reward": -5.06103515625,
"reward_std": 5.579965591430664,
"rewards/rm_reward_func/mean": -5.06103515625,
"rewards/rm_reward_func/std": 10.378743171691895,
"step": 1150
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 484.0,
"completions/mean_length": 267.84375,
"completions/mean_terminated_length": 222.629638671875,
"completions/min_length": 58.0,
"completions/min_terminated_length": 58.0,
"epoch": 0.9208,
"grad_norm": 25.507362365722656,
"kl": 0.8896484375,
"learning_rate": 1e-06,
"loss": -0.1392,
"num_tokens": 15135672.0,
"reward": 1.32666015625,
"reward_std": 8.384624481201172,
"rewards/rm_reward_func/mean": 1.32666015625,
"rewards/rm_reward_func/std": 11.488581657409668,
"step": 1151
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 507.0,
"completions/mean_length": 356.6875,
"completions/mean_terminated_length": 263.5,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9216,
"grad_norm": 9.581064224243164,
"kl": 1.3369140625,
"learning_rate": 1e-06,
"loss": 0.037,
"num_tokens": 15150006.0,
"reward": -1.42645263671875,
"reward_std": 4.21787691116333,
"rewards/rm_reward_func/mean": -1.42645263671875,
"rewards/rm_reward_func/std": 8.071844100952148,
"step": 1152
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 506.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 222.625,
"completions/mean_terminated_length": 222.625,
"completions/min_length": 118.0,
"completions/min_terminated_length": 118.0,
"epoch": 0.9224,
"grad_norm": 18.28358268737793,
"kl": 0.68798828125,
"learning_rate": 1e-06,
"loss": 0.0422,
"num_tokens": 15159922.0,
"reward": -3.984771728515625,
"reward_std": 4.933001518249512,
"rewards/rm_reward_func/mean": -3.984771728515625,
"rewards/rm_reward_func/std": 5.989495277404785,
"step": 1153
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 261.90625,
"completions/mean_terminated_length": 253.8386993408203,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9232,
"grad_norm": 16.853899002075195,
"kl": 1.552734375,
"learning_rate": 1e-06,
"loss": -0.0088,
"num_tokens": 15170879.0,
"reward": 4.74188232421875,
"reward_std": 8.369260787963867,
"rewards/rm_reward_func/mean": 4.74188232421875,
"rewards/rm_reward_func/std": 15.880949974060059,
"step": 1154
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 496.0,
"completions/mean_length": 277.875,
"completions/mean_terminated_length": 223.84616088867188,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.924,
"grad_norm": 8.054681777954102,
"kl": 1.3564453125,
"learning_rate": 1e-06,
"loss": 0.0452,
"num_tokens": 15182435.0,
"reward": -0.05224609375,
"reward_std": 5.243035316467285,
"rewards/rm_reward_func/mean": -0.05224609375,
"rewards/rm_reward_func/std": 11.808974266052246,
"step": 1155
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 483.0,
"completions/mean_length": 370.6875,
"completions/mean_terminated_length": 331.1199951171875,
"completions/min_length": 148.0,
"completions/min_terminated_length": 148.0,
"epoch": 0.9248,
"grad_norm": 12.995421409606934,
"kl": 0.4638671875,
"learning_rate": 1e-06,
"loss": 0.075,
"num_tokens": 15196273.0,
"reward": -0.77685546875,
"reward_std": 3.534580945968628,
"rewards/rm_reward_func/mean": -0.77685546875,
"rewards/rm_reward_func/std": 10.869222640991211,
"step": 1156
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 504.0,
"completions/mean_length": 375.25,
"completions/mean_terminated_length": 303.6190490722656,
"completions/min_length": 52.0,
"completions/min_terminated_length": 52.0,
"epoch": 0.9256,
"grad_norm": 29.573589324951172,
"kl": 2.072265625,
"learning_rate": 1e-06,
"loss": 0.0377,
"num_tokens": 15210377.0,
"reward": 3.455078125,
"reward_std": 9.187593460083008,
"rewards/rm_reward_func/mean": 3.455078125,
"rewards/rm_reward_func/std": 15.553760528564453,
"step": 1157
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 482.0,
"completions/mean_length": 393.75,
"completions/mean_terminated_length": 347.478271484375,
"completions/min_length": 149.0,
"completions/min_terminated_length": 149.0,
"epoch": 0.9264,
"grad_norm": 10.407255172729492,
"kl": 1.662109375,
"learning_rate": 1e-06,
"loss": 0.0592,
"num_tokens": 15225153.0,
"reward": -4.851356506347656,
"reward_std": 8.531854629516602,
"rewards/rm_reward_func/mean": -4.851356506347656,
"rewards/rm_reward_func/std": 12.067245483398438,
"step": 1158
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 483.0,
"completions/max_terminated_length": 483.0,
"completions/mean_length": 228.5,
"completions/mean_terminated_length": 228.5,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9272,
"grad_norm": 22.906204223632812,
"kl": 0.4560546875,
"learning_rate": 1e-06,
"loss": -0.2539,
"num_tokens": 15236201.0,
"reward": 4.116214752197266,
"reward_std": 7.407968521118164,
"rewards/rm_reward_func/mean": 4.116214752197266,
"rewards/rm_reward_func/std": 12.722545623779297,
"step": 1159
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 492.0,
"completions/mean_length": 351.09375,
"completions/mean_terminated_length": 313.9615478515625,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.928,
"grad_norm": 17.148536682128906,
"kl": 0.40380859375,
"learning_rate": 1e-06,
"loss": 0.0482,
"num_tokens": 15250796.0,
"reward": 10.2752685546875,
"reward_std": 4.209094047546387,
"rewards/rm_reward_func/mean": 10.2752685546875,
"rewards/rm_reward_func/std": 11.812329292297363,
"step": 1160
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 407.0,
"completions/max_terminated_length": 407.0,
"completions/mean_length": 171.5,
"completions/mean_terminated_length": 171.5,
"completions/min_length": 27.0,
"completions/min_terminated_length": 27.0,
"epoch": 0.9288,
"grad_norm": 42.1457633972168,
"kl": 1.83056640625,
"learning_rate": 1e-06,
"loss": 0.2306,
"num_tokens": 15258540.0,
"reward": 0.6884765625,
"reward_std": 7.095283508300781,
"rewards/rm_reward_func/mean": 0.6884765625,
"rewards/rm_reward_func/std": 11.873205184936523,
"step": 1161
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 511.0,
"completions/mean_length": 369.46875,
"completions/mean_terminated_length": 313.6956481933594,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9296,
"grad_norm": 7.2761640548706055,
"kl": 2.3427734375,
"learning_rate": 1e-06,
"loss": 0.0776,
"num_tokens": 15272827.0,
"reward": 9.095703125,
"reward_std": 6.575770378112793,
"rewards/rm_reward_func/mean": 9.095703125,
"rewards/rm_reward_func/std": 15.32244873046875,
"step": 1162
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 497.0,
"completions/max_terminated_length": 497.0,
"completions/mean_length": 166.8125,
"completions/mean_terminated_length": 166.8125,
"completions/min_length": 73.0,
"completions/min_terminated_length": 73.0,
"epoch": 0.9304,
"grad_norm": 10.371278762817383,
"kl": 0.509765625,
"learning_rate": 1e-06,
"loss": -0.0377,
"num_tokens": 15280869.0,
"reward": 5.901123046875,
"reward_std": 2.3234264850616455,
"rewards/rm_reward_func/mean": 5.901123046875,
"rewards/rm_reward_func/std": 10.951936721801758,
"step": 1163
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 499.0,
"completions/mean_length": 296.21875,
"completions/mean_terminated_length": 211.78260803222656,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9312,
"grad_norm": 7.607293605804443,
"kl": 2.140625,
"learning_rate": 1e-06,
"loss": -0.0091,
"num_tokens": 15292900.0,
"reward": -0.714080810546875,
"reward_std": 6.104308128356934,
"rewards/rm_reward_func/mean": -0.714080810546875,
"rewards/rm_reward_func/std": 9.522336959838867,
"step": 1164
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 332.71875,
"completions/mean_terminated_length": 291.3461608886719,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.932,
"grad_norm": 6.345226764678955,
"kl": 1.32861328125,
"learning_rate": 1e-06,
"loss": 0.0391,
"num_tokens": 15306323.0,
"reward": 9.87939453125,
"reward_std": 8.231972694396973,
"rewards/rm_reward_func/mean": 9.87939453125,
"rewards/rm_reward_func/std": 16.97296142578125,
"step": 1165
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 296.71875,
"completions/mean_terminated_length": 274.4482727050781,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9328,
"grad_norm": 11.910536766052246,
"kl": 3.41845703125,
"learning_rate": 1e-06,
"loss": -0.0196,
"num_tokens": 15318378.0,
"reward": -2.5654296875,
"reward_std": 8.650764465332031,
"rewards/rm_reward_func/mean": -2.5654296875,
"rewards/rm_reward_func/std": 14.586143493652344,
"step": 1166
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 370.0625,
"completions/mean_terminated_length": 337.3077087402344,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9336,
"grad_norm": 5.885568618774414,
"kl": 0.9931640625,
"learning_rate": 1e-06,
"loss": 0.0021,
"num_tokens": 15334220.0,
"reward": 14.67633056640625,
"reward_std": 8.415434837341309,
"rewards/rm_reward_func/mean": 14.67633056640625,
"rewards/rm_reward_func/std": 19.035816192626953,
"step": 1167
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 492.0,
"completions/mean_length": 397.21875,
"completions/mean_terminated_length": 389.5666809082031,
"completions/min_length": 167.0,
"completions/min_terminated_length": 167.0,
"epoch": 0.9344,
"grad_norm": 11.35747241973877,
"kl": 2.22021484375,
"learning_rate": 1e-06,
"loss": 0.0588,
"num_tokens": 15349371.0,
"reward": 7.377197265625,
"reward_std": 6.683224678039551,
"rewards/rm_reward_func/mean": 7.377197265625,
"rewards/rm_reward_func/std": 13.624530792236328,
"step": 1168
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 476.0,
"completions/mean_length": 369.375,
"completions/mean_terminated_length": 349.0000305175781,
"completions/min_length": 157.0,
"completions/min_terminated_length": 157.0,
"epoch": 0.9352,
"grad_norm": 28.526676177978516,
"kl": 4.8515625,
"learning_rate": 1e-06,
"loss": 0.1177,
"num_tokens": 15363063.0,
"reward": 3.8070831298828125,
"reward_std": 14.380706787109375,
"rewards/rm_reward_func/mean": 3.8070831298828125,
"rewards/rm_reward_func/std": 15.08162784576416,
"step": 1169
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 500.0,
"completions/max_terminated_length": 500.0,
"completions/mean_length": 277.78125,
"completions/mean_terminated_length": 277.78125,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.936,
"grad_norm": 7.368760585784912,
"kl": 0.72314453125,
"learning_rate": 1e-06,
"loss": 0.0015,
"num_tokens": 15373880.0,
"reward": 2.238525390625,
"reward_std": 2.461265802383423,
"rewards/rm_reward_func/mean": 2.238525390625,
"rewards/rm_reward_func/std": 6.371425628662109,
"step": 1170
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 503.0,
"completions/mean_length": 371.5,
"completions/mean_terminated_length": 324.66668701171875,
"completions/min_length": 94.0,
"completions/min_terminated_length": 94.0,
"epoch": 0.9368,
"grad_norm": 23.937978744506836,
"kl": 2.81396484375,
"learning_rate": 1e-06,
"loss": 0.0015,
"num_tokens": 15387984.0,
"reward": -3.2255859375,
"reward_std": 8.331253051757812,
"rewards/rm_reward_func/mean": -3.2255859375,
"rewards/rm_reward_func/std": 10.779619216918945,
"step": 1171
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 487.0,
"completions/mean_length": 378.65625,
"completions/mean_terminated_length": 318.04547119140625,
"completions/min_length": 103.0,
"completions/min_terminated_length": 103.0,
"epoch": 0.9376,
"grad_norm": 15.15499496459961,
"kl": 4.59619140625,
"learning_rate": 1e-06,
"loss": 0.0643,
"num_tokens": 15402765.0,
"reward": -3.593994140625,
"reward_std": 6.05653190612793,
"rewards/rm_reward_func/mean": -3.593994140625,
"rewards/rm_reward_func/std": 10.450895309448242,
"step": 1172
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 506.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 241.40625,
"completions/mean_terminated_length": 241.40625,
"completions/min_length": 72.0,
"completions/min_terminated_length": 72.0,
"epoch": 0.9384,
"grad_norm": 10.372428894042969,
"kl": 2.1416015625,
"learning_rate": 1e-06,
"loss": 0.0091,
"num_tokens": 15417634.0,
"reward": 6.179931640625,
"reward_std": 7.2046685218811035,
"rewards/rm_reward_func/mean": 6.179931640625,
"rewards/rm_reward_func/std": 8.590556144714355,
"step": 1173
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 276.90625,
"completions/mean_terminated_length": 222.6538543701172,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9392,
"grad_norm": 14.381688117980957,
"kl": 1.92138671875,
"learning_rate": 1e-06,
"loss": 0.0851,
"num_tokens": 15429623.0,
"reward": -2.0927734375,
"reward_std": 4.637411117553711,
"rewards/rm_reward_func/mean": -2.0927734375,
"rewards/rm_reward_func/std": 13.167999267578125,
"step": 1174
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 501.0,
"completions/mean_length": 280.53125,
"completions/mean_terminated_length": 265.1000061035156,
"completions/min_length": 122.0,
"completions/min_terminated_length": 122.0,
"epoch": 0.94,
"grad_norm": 32.04814529418945,
"kl": 7.716796875,
"learning_rate": 1e-06,
"loss": 0.1969,
"num_tokens": 15440720.0,
"reward": -1.5761604309082031,
"reward_std": 14.394807815551758,
"rewards/rm_reward_func/mean": -1.5761604309082031,
"rewards/rm_reward_func/std": 17.306682586669922,
"step": 1175
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 365.46875,
"completions/mean_terminated_length": 331.65386962890625,
"completions/min_length": 51.0,
"completions/min_terminated_length": 51.0,
"epoch": 0.9408,
"grad_norm": 6.667887210845947,
"kl": 1.29052734375,
"learning_rate": 1e-06,
"loss": -0.0015,
"num_tokens": 15454551.0,
"reward": 3.3128662109375,
"reward_std": 7.290994644165039,
"rewards/rm_reward_func/mean": 3.3128662109375,
"rewards/rm_reward_func/std": 11.585569381713867,
"step": 1176
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 504.0,
"completions/mean_length": 372.25,
"completions/mean_terminated_length": 317.5652160644531,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9416,
"grad_norm": 6.514392375946045,
"kl": 3.54052734375,
"learning_rate": 1e-06,
"loss": 0.1126,
"num_tokens": 15469031.0,
"reward": 15.22454833984375,
"reward_std": 8.186134338378906,
"rewards/rm_reward_func/mean": 15.22454833984375,
"rewards/rm_reward_func/std": 20.967836380004883,
"step": 1177
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 352.0,
"completions/max_terminated_length": 352.0,
"completions/mean_length": 128.28125,
"completions/mean_terminated_length": 128.28125,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9424,
"grad_norm": 57.319156646728516,
"kl": 0.97265625,
"learning_rate": 1e-06,
"loss": 0.0197,
"num_tokens": 15475856.0,
"reward": 4.0428466796875,
"reward_std": 1.0468058586120605,
"rewards/rm_reward_func/mean": 4.0428466796875,
"rewards/rm_reward_func/std": 8.96422004699707,
"step": 1178
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 466.0,
"completions/max_terminated_length": 466.0,
"completions/mean_length": 257.375,
"completions/mean_terminated_length": 257.375,
"completions/min_length": 38.0,
"completions/min_terminated_length": 38.0,
"epoch": 0.9432,
"grad_norm": 37.18678283691406,
"kl": 4.50732421875,
"learning_rate": 1e-06,
"loss": 0.1144,
"num_tokens": 15490860.0,
"reward": -0.7977094650268555,
"reward_std": 9.967428207397461,
"rewards/rm_reward_func/mean": -0.7977094650268555,
"rewards/rm_reward_func/std": 12.24376106262207,
"step": 1179
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 498.0,
"completions/mean_length": 274.75,
"completions/mean_terminated_length": 208.3199920654297,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.944,
"grad_norm": 11.060836791992188,
"kl": 2.35205078125,
"learning_rate": 1e-06,
"loss": 0.0886,
"num_tokens": 15502980.0,
"reward": 4.240234375,
"reward_std": 2.008146047592163,
"rewards/rm_reward_func/mean": 4.240234375,
"rewards/rm_reward_func/std": 12.89011287689209,
"step": 1180
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 487.0,
"completions/max_terminated_length": 487.0,
"completions/mean_length": 224.71875,
"completions/mean_terminated_length": 224.71875,
"completions/min_length": 55.0,
"completions/min_terminated_length": 55.0,
"epoch": 0.9448,
"grad_norm": 25.126611709594727,
"kl": 3.6806640625,
"learning_rate": 1e-06,
"loss": 0.0654,
"num_tokens": 15512067.0,
"reward": -1.59136962890625,
"reward_std": 6.693836212158203,
"rewards/rm_reward_func/mean": -1.59136962890625,
"rewards/rm_reward_func/std": 13.480401039123535,
"step": 1181
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 502.0,
"completions/mean_length": 310.40625,
"completions/mean_terminated_length": 296.9666748046875,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9456,
"grad_norm": 6.820868968963623,
"kl": 2.0498046875,
"learning_rate": 1e-06,
"loss": 0.0904,
"num_tokens": 15527256.0,
"reward": 11.24267578125,
"reward_std": 6.773592472076416,
"rewards/rm_reward_func/mean": 11.24267578125,
"rewards/rm_reward_func/std": 16.010009765625,
"step": 1182
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 324.3125,
"completions/mean_terminated_length": 271.7599792480469,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9464,
"grad_norm": 5.817383289337158,
"kl": 1.25830078125,
"learning_rate": 1e-06,
"loss": 0.0173,
"num_tokens": 15540034.0,
"reward": 12.9453125,
"reward_std": 6.853506088256836,
"rewards/rm_reward_func/mean": 12.9453125,
"rewards/rm_reward_func/std": 11.050248146057129,
"step": 1183
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.59375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 394.96875,
"completions/mean_terminated_length": 223.92308044433594,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9472,
"grad_norm": 11.657125473022461,
"kl": 1.68115234375,
"learning_rate": 1e-06,
"loss": 0.0629,
"num_tokens": 15556841.0,
"reward": 1.7704925537109375,
"reward_std": 4.207978248596191,
"rewards/rm_reward_func/mean": 1.7704925537109375,
"rewards/rm_reward_func/std": 6.741135120391846,
"step": 1184
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 461.0,
"completions/mean_length": 329.9375,
"completions/mean_terminated_length": 303.9285888671875,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.948,
"grad_norm": 26.10344696044922,
"kl": 4.341796875,
"learning_rate": 1e-06,
"loss": 0.2526,
"num_tokens": 15570263.0,
"reward": 7.865570068359375,
"reward_std": 6.319413185119629,
"rewards/rm_reward_func/mean": 7.865570068359375,
"rewards/rm_reward_func/std": 15.726142883300781,
"step": 1185
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 461.0,
"completions/mean_length": 335.1875,
"completions/mean_terminated_length": 276.25,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9488,
"grad_norm": 6.295200824737549,
"kl": 0.4013671875,
"learning_rate": 1e-06,
"loss": 0.0074,
"num_tokens": 15584085.0,
"reward": 7.826171875,
"reward_std": 1.844343900680542,
"rewards/rm_reward_func/mean": 7.826171875,
"rewards/rm_reward_func/std": 16.900344848632812,
"step": 1186
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 414.0,
"completions/mean_length": 311.96875,
"completions/mean_terminated_length": 255.95999145507812,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9496,
"grad_norm": 7.316650867462158,
"kl": 0.916015625,
"learning_rate": 1e-06,
"loss": -0.0087,
"num_tokens": 15598652.0,
"reward": -1.8782958984375,
"reward_std": 4.917675495147705,
"rewards/rm_reward_func/mean": -1.8782958984375,
"rewards/rm_reward_func/std": 8.429694175720215,
"step": 1187
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 484.0,
"completions/mean_length": 367.78125,
"completions/mean_terminated_length": 358.16668701171875,
"completions/min_length": 204.0,
"completions/min_terminated_length": 204.0,
"epoch": 0.9504,
"grad_norm": 27.917741775512695,
"kl": 1.7177734375,
"learning_rate": 1e-06,
"loss": 0.0232,
"num_tokens": 15612501.0,
"reward": 11.6102294921875,
"reward_std": 8.347414016723633,
"rewards/rm_reward_func/mean": 11.6102294921875,
"rewards/rm_reward_func/std": 13.412553787231445,
"step": 1188
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 443.46875,
"completions/mean_terminated_length": 420.625,
"completions/min_length": 184.0,
"completions/min_terminated_length": 184.0,
"epoch": 0.9512,
"grad_norm": 19.72421646118164,
"kl": 3.552734375,
"learning_rate": 1e-06,
"loss": 0.1278,
"num_tokens": 15628892.0,
"reward": 5.533203125,
"reward_std": 10.962414741516113,
"rewards/rm_reward_func/mean": 5.533203125,
"rewards/rm_reward_func/std": 24.766939163208008,
"step": 1189
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 414.0,
"completions/mean_length": 296.40625,
"completions/mean_terminated_length": 289.45159912109375,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.952,
"grad_norm": 8.21200180053711,
"kl": 1.095703125,
"learning_rate": 1e-06,
"loss": 0.0381,
"num_tokens": 15642537.0,
"reward": 6.253631591796875,
"reward_std": 4.122445106506348,
"rewards/rm_reward_func/mean": 6.253631591796875,
"rewards/rm_reward_func/std": 6.488741874694824,
"step": 1190
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 486.0,
"completions/mean_length": 342.65625,
"completions/mean_terminated_length": 331.3666687011719,
"completions/min_length": 97.0,
"completions/min_terminated_length": 97.0,
"epoch": 0.9528,
"grad_norm": 7.0987677574157715,
"kl": 2.54150390625,
"learning_rate": 1e-06,
"loss": 0.0774,
"num_tokens": 15657294.0,
"reward": 4.48828125,
"reward_std": 5.943986892700195,
"rewards/rm_reward_func/mean": 4.48828125,
"rewards/rm_reward_func/std": 18.430313110351562,
"step": 1191
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 337.0625,
"completions/mean_terminated_length": 296.69232177734375,
"completions/min_length": 137.0,
"completions/min_terminated_length": 137.0,
"epoch": 0.9536,
"grad_norm": 10.321151733398438,
"kl": 0.74853515625,
"learning_rate": 1e-06,
"loss": 0.0207,
"num_tokens": 15670624.0,
"reward": 8.99795150756836,
"reward_std": 6.07137393951416,
"rewards/rm_reward_func/mean": 8.99795150756836,
"rewards/rm_reward_func/std": 11.28697395324707,
"step": 1192
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 511.0,
"completions/mean_length": 406.46875,
"completions/mean_terminated_length": 365.1739196777344,
"completions/min_length": 156.0,
"completions/min_terminated_length": 156.0,
"epoch": 0.9544,
"grad_norm": 75.35392761230469,
"kl": 8.7998046875,
"learning_rate": 1e-06,
"loss": 0.3095,
"num_tokens": 15688439.0,
"reward": -8.990234375,
"reward_std": 6.966933250427246,
"rewards/rm_reward_func/mean": -8.990234375,
"rewards/rm_reward_func/std": 18.05649757385254,
"step": 1193
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 239.8125,
"completions/mean_terminated_length": 211.65516662597656,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9552,
"grad_norm": 15.314480781555176,
"kl": 1.806640625,
"learning_rate": 1e-06,
"loss": 0.0248,
"num_tokens": 15700577.0,
"reward": 3.8753662109375,
"reward_std": 4.899765491485596,
"rewards/rm_reward_func/mean": 3.8753662109375,
"rewards/rm_reward_func/std": 8.12557315826416,
"step": 1194
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 362.15625,
"completions/mean_terminated_length": 327.5769348144531,
"completions/min_length": 149.0,
"completions/min_terminated_length": 149.0,
"epoch": 0.956,
"grad_norm": 78.24580383300781,
"kl": 8.9765625,
"learning_rate": 1e-06,
"loss": 0.2552,
"num_tokens": 15714486.0,
"reward": -6.42059326171875,
"reward_std": 9.736968994140625,
"rewards/rm_reward_func/mean": -6.42059326171875,
"rewards/rm_reward_func/std": 14.20551586151123,
"step": 1195
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.28125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 397.0,
"completions/mean_length": 280.46875,
"completions/mean_terminated_length": 189.86956787109375,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9568,
"grad_norm": 31.360624313354492,
"kl": 3.84814453125,
"learning_rate": 1e-06,
"loss": 0.19,
"num_tokens": 15726301.0,
"reward": 1.2685317993164062,
"reward_std": 4.518694877624512,
"rewards/rm_reward_func/mean": 1.2685317993164062,
"rewards/rm_reward_func/std": 8.930562973022461,
"step": 1196
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 274.03125,
"completions/mean_terminated_length": 240.0357208251953,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9576,
"grad_norm": 7.44788122177124,
"kl": 0.564453125,
"learning_rate": 1e-06,
"loss": 0.0323,
"num_tokens": 15738126.0,
"reward": 8.062255859375,
"reward_std": 1.784210443496704,
"rewards/rm_reward_func/mean": 8.062255859375,
"rewards/rm_reward_func/std": 6.24770450592041,
"step": 1197
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 430.0,
"completions/max_terminated_length": 430.0,
"completions/mean_length": 200.1875,
"completions/mean_terminated_length": 200.1875,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9584,
"grad_norm": 46.06965637207031,
"kl": 6.3017578125,
"learning_rate": 1e-06,
"loss": 0.2246,
"num_tokens": 15746660.0,
"reward": -1.352294921875,
"reward_std": 7.574354648590088,
"rewards/rm_reward_func/mean": -1.352294921875,
"rewards/rm_reward_func/std": 14.081011772155762,
"step": 1198
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 487.0,
"completions/mean_length": 303.0625,
"completions/mean_terminated_length": 264.370361328125,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9592,
"grad_norm": 13.53447437286377,
"kl": 3.17578125,
"learning_rate": 1e-06,
"loss": 0.0633,
"num_tokens": 15761318.0,
"reward": 6.076904296875,
"reward_std": 8.05277156829834,
"rewards/rm_reward_func/mean": 6.076904296875,
"rewards/rm_reward_func/std": 10.448375701904297,
"step": 1199
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 502.0,
"completions/mean_length": 372.71875,
"completions/mean_terminated_length": 358.3103332519531,
"completions/min_length": 108.0,
"completions/min_terminated_length": 108.0,
"epoch": 0.96,
"grad_norm": 82.56756591796875,
"kl": 6.08251953125,
"learning_rate": 1e-06,
"loss": 0.1814,
"num_tokens": 15777317.0,
"reward": 0.806884765625,
"reward_std": 8.0141019821167,
"rewards/rm_reward_func/mean": 0.806884765625,
"rewards/rm_reward_func/std": 20.149206161499023,
"step": 1200
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 80.0,
"completions/mean_length": 134.0,
"completions/mean_terminated_length": 80.0,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9608,
"grad_norm": 80.49437713623047,
"kl": 2.533203125,
"learning_rate": 1e-06,
"loss": 0.3195,
"num_tokens": 15786189.0,
"reward": 2.772796630859375,
"reward_std": 2.7184181213378906,
"rewards/rm_reward_func/mean": 2.772796630859375,
"rewards/rm_reward_func/std": 8.184410095214844,
"step": 1201
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 501.0,
"completions/mean_length": 355.71875,
"completions/mean_terminated_length": 326.77777099609375,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9616,
"grad_norm": 22.120071411132812,
"kl": 0.51708984375,
"learning_rate": 1e-06,
"loss": 0.0185,
"num_tokens": 15802404.0,
"reward": 19.8399658203125,
"reward_std": 4.591503620147705,
"rewards/rm_reward_func/mean": 19.8399658203125,
"rewards/rm_reward_func/std": 21.975711822509766,
"step": 1202
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 399.125,
"completions/mean_terminated_length": 383.0000305175781,
"completions/min_length": 156.0,
"completions/min_terminated_length": 156.0,
"epoch": 0.9624,
"grad_norm": 13.387299537658691,
"kl": 4.68603515625,
"learning_rate": 1e-06,
"loss": 0.2318,
"num_tokens": 15818008.0,
"reward": 5.8818359375,
"reward_std": 6.727931976318359,
"rewards/rm_reward_func/mean": 5.8818359375,
"rewards/rm_reward_func/std": 23.548294067382812,
"step": 1203
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 284.125,
"completions/mean_terminated_length": 241.92593383789062,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9632,
"grad_norm": 5.043156147003174,
"kl": 2.41845703125,
"learning_rate": 1e-06,
"loss": 0.096,
"num_tokens": 15830828.0,
"reward": 2.5513916015625,
"reward_std": 6.220531463623047,
"rewards/rm_reward_func/mean": 2.5513916015625,
"rewards/rm_reward_func/std": 13.680570602416992,
"step": 1204
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 477.0,
"completions/mean_length": 379.25,
"completions/mean_terminated_length": 365.5172424316406,
"completions/min_length": 177.0,
"completions/min_terminated_length": 177.0,
"epoch": 0.964,
"grad_norm": 19.86156463623047,
"kl": 4.646484375,
"learning_rate": 1e-06,
"loss": 0.19,
"num_tokens": 15844940.0,
"reward": 12.568359375,
"reward_std": 12.276927947998047,
"rewards/rm_reward_func/mean": 12.568359375,
"rewards/rm_reward_func/std": 18.882339477539062,
"step": 1205
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 472.0,
"completions/mean_length": 244.0625,
"completions/mean_terminated_length": 226.20001220703125,
"completions/min_length": 29.0,
"completions/min_terminated_length": 29.0,
"epoch": 0.9648,
"grad_norm": 35.16064453125,
"kl": 3.6748046875,
"learning_rate": 1e-06,
"loss": -0.0097,
"num_tokens": 15854710.0,
"reward": 4.755615234375,
"reward_std": 10.025222778320312,
"rewards/rm_reward_func/mean": 4.755615234375,
"rewards/rm_reward_func/std": 17.34681510925293,
"step": 1206
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 480.0,
"completions/mean_length": 330.71875,
"completions/mean_terminated_length": 297.1481628417969,
"completions/min_length": 148.0,
"completions/min_terminated_length": 148.0,
"epoch": 0.9656,
"grad_norm": 20.283857345581055,
"kl": 4.10986328125,
"learning_rate": 1e-06,
"loss": 0.1873,
"num_tokens": 15869749.0,
"reward": -1.422607421875,
"reward_std": 7.128761291503906,
"rewards/rm_reward_func/mean": -1.422607421875,
"rewards/rm_reward_func/std": 8.89996337890625,
"step": 1207
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 461.0,
"completions/max_terminated_length": 461.0,
"completions/mean_length": 228.8125,
"completions/mean_terminated_length": 228.8125,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9664,
"grad_norm": 14.01342487335205,
"kl": 1.6328125,
"learning_rate": 1e-06,
"loss": 0.0612,
"num_tokens": 15879927.0,
"reward": 5.265869140625,
"reward_std": 2.066286087036133,
"rewards/rm_reward_func/mean": 5.265869140625,
"rewards/rm_reward_func/std": 7.653271675109863,
"step": 1208
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 509.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 301.34375,
"completions/mean_terminated_length": 301.34375,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9672,
"grad_norm": 16.027812957763672,
"kl": 2.18359375,
"learning_rate": 1e-06,
"loss": -0.007,
"num_tokens": 15895018.0,
"reward": 5.47796630859375,
"reward_std": 9.107280731201172,
"rewards/rm_reward_func/mean": 5.47796630859375,
"rewards/rm_reward_func/std": 13.736273765563965,
"step": 1209
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 452.0,
"completions/mean_length": 366.75,
"completions/mean_terminated_length": 326.0799865722656,
"completions/min_length": 28.0,
"completions/min_terminated_length": 28.0,
"epoch": 0.968,
"grad_norm": 6.4775848388671875,
"kl": 0.8310546875,
"learning_rate": 1e-06,
"loss": -0.1347,
"num_tokens": 15909162.0,
"reward": 4.66473388671875,
"reward_std": 10.003836631774902,
"rewards/rm_reward_func/mean": 4.66473388671875,
"rewards/rm_reward_func/std": 11.560744285583496,
"step": 1210
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 442.0,
"completions/max_terminated_length": 442.0,
"completions/mean_length": 250.40625,
"completions/mean_terminated_length": 250.40625,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9688,
"grad_norm": 9.306504249572754,
"kl": 1.1865234375,
"learning_rate": 1e-06,
"loss": 0.0649,
"num_tokens": 15920199.0,
"reward": 11.21533203125,
"reward_std": 4.61986780166626,
"rewards/rm_reward_func/mean": 11.21533203125,
"rewards/rm_reward_func/std": 11.726616859436035,
"step": 1211
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 428.0,
"completions/mean_length": 276.8125,
"completions/mean_terminated_length": 243.21429443359375,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9696,
"grad_norm": 29.228607177734375,
"kl": 1.453125,
"learning_rate": 1e-06,
"loss": 0.0867,
"num_tokens": 15932337.0,
"reward": -3.7728271484375,
"reward_std": 2.3056063652038574,
"rewards/rm_reward_func/mean": -3.7728271484375,
"rewards/rm_reward_func/std": 5.636190891265869,
"step": 1212
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 476.0,
"completions/mean_length": 242.71875,
"completions/mean_terminated_length": 204.25001525878906,
"completions/min_length": 78.0,
"completions/min_terminated_length": 78.0,
"epoch": 0.9704,
"grad_norm": 28.853511810302734,
"kl": 2.189453125,
"learning_rate": 1e-06,
"loss": 0.0766,
"num_tokens": 15942880.0,
"reward": 4.451416015625,
"reward_std": 4.814798355102539,
"rewards/rm_reward_func/mean": 4.451416015625,
"rewards/rm_reward_func/std": 7.7672343254089355,
"step": 1213
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 490.0,
"completions/mean_length": 307.4375,
"completions/mean_terminated_length": 260.23077392578125,
"completions/min_length": 63.0,
"completions/min_terminated_length": 63.0,
"epoch": 0.9712,
"grad_norm": 8.984874725341797,
"kl": 0.439697265625,
"learning_rate": 1e-06,
"loss": 0.0709,
"num_tokens": 15954926.0,
"reward": 2.17816162109375,
"reward_std": 5.897883415222168,
"rewards/rm_reward_func/mean": 2.17816162109375,
"rewards/rm_reward_func/std": 6.458736419677734,
"step": 1214
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 421.0,
"completions/max_terminated_length": 421.0,
"completions/mean_length": 164.15625,
"completions/mean_terminated_length": 164.15625,
"completions/min_length": 77.0,
"completions/min_terminated_length": 77.0,
"epoch": 0.972,
"grad_norm": 15.127213478088379,
"kl": 0.443115234375,
"learning_rate": 1e-06,
"loss": 0.0297,
"num_tokens": 15965971.0,
"reward": 8.07522964477539,
"reward_std": 1.5865893363952637,
"rewards/rm_reward_func/mean": 8.07522964477539,
"rewards/rm_reward_func/std": 9.159050941467285,
"step": 1215
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 502.0,
"completions/mean_length": 255.9375,
"completions/mean_terminated_length": 247.6774139404297,
"completions/min_length": 71.0,
"completions/min_terminated_length": 71.0,
"epoch": 0.9728,
"grad_norm": 42.37367630004883,
"kl": 2.08447265625,
"learning_rate": 1e-06,
"loss": 0.0359,
"num_tokens": 15979769.0,
"reward": 1.78619384765625,
"reward_std": 2.9838078022003174,
"rewards/rm_reward_func/mean": 1.78619384765625,
"rewards/rm_reward_func/std": 11.067867279052734,
"step": 1216
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 483.0,
"completions/mean_length": 346.59375,
"completions/mean_terminated_length": 271.4090881347656,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9736,
"grad_norm": 9.83594799041748,
"kl": 4.3447265625,
"learning_rate": 1e-06,
"loss": 0.1791,
"num_tokens": 15996100.0,
"reward": -1.2491302490234375,
"reward_std": 3.7429025173187256,
"rewards/rm_reward_func/mean": -1.2491302490234375,
"rewards/rm_reward_func/std": 8.419785499572754,
"step": 1217
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 487.0,
"completions/mean_length": 353.59375,
"completions/mean_terminated_length": 343.0333557128906,
"completions/min_length": 100.0,
"completions/min_terminated_length": 100.0,
"epoch": 0.9744,
"grad_norm": 10.141213417053223,
"kl": 2.3017578125,
"learning_rate": 1e-06,
"loss": 0.0357,
"num_tokens": 16009191.0,
"reward": 3.6845703125,
"reward_std": 9.749015808105469,
"rewards/rm_reward_func/mean": 3.6845703125,
"rewards/rm_reward_func/std": 15.83315658569336,
"step": 1218
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 501.0,
"completions/mean_length": 451.40625,
"completions/mean_terminated_length": 419.66668701171875,
"completions/min_length": 304.0,
"completions/min_terminated_length": 304.0,
"epoch": 0.9752,
"grad_norm": 6.541053771972656,
"kl": 0.66943359375,
"learning_rate": 1e-06,
"loss": 0.0511,
"num_tokens": 16026156.0,
"reward": 10.817626953125,
"reward_std": 10.195655822753906,
"rewards/rm_reward_func/mean": 10.817626953125,
"rewards/rm_reward_func/std": 14.443761825561523,
"step": 1219
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 509.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 308.6875,
"completions/mean_terminated_length": 308.6875,
"completions/min_length": 65.0,
"completions/min_terminated_length": 65.0,
"epoch": 0.976,
"grad_norm": 7.883213996887207,
"kl": 0.72705078125,
"learning_rate": 1e-06,
"loss": -0.0421,
"num_tokens": 16040322.0,
"reward": 2.47705078125,
"reward_std": 7.948477745056152,
"rewards/rm_reward_func/mean": 2.47705078125,
"rewards/rm_reward_func/std": 14.740900039672852,
"step": 1220
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 231.6875,
"completions/mean_terminated_length": 231.6875,
"completions/min_length": 40.0,
"completions/min_terminated_length": 40.0,
"epoch": 0.9768,
"grad_norm": 16.915742874145508,
"kl": 1.87890625,
"learning_rate": 1e-06,
"loss": 0.0358,
"num_tokens": 16050816.0,
"reward": -4.7006988525390625,
"reward_std": 6.505680561065674,
"rewards/rm_reward_func/mean": -4.7006988525390625,
"rewards/rm_reward_func/std": 9.603677749633789,
"step": 1221
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 248.0,
"completions/mean_terminated_length": 210.2857208251953,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9776,
"grad_norm": 7.012750625610352,
"kl": 0.8623046875,
"learning_rate": 1e-06,
"loss": 0.0782,
"num_tokens": 16061744.0,
"reward": 12.907470703125,
"reward_std": 4.333094596862793,
"rewards/rm_reward_func/mean": 12.907470703125,
"rewards/rm_reward_func/std": 8.020941734313965,
"step": 1222
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 401.0,
"completions/max_terminated_length": 401.0,
"completions/mean_length": 228.8125,
"completions/mean_terminated_length": 228.8125,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9784,
"grad_norm": 6.7300801277160645,
"kl": 0.89794921875,
"learning_rate": 1e-06,
"loss": 0.0436,
"num_tokens": 16072642.0,
"reward": -0.694091796875,
"reward_std": 3.24855637550354,
"rewards/rm_reward_func/mean": -0.694091796875,
"rewards/rm_reward_func/std": 10.3416166305542,
"step": 1223
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 339.0,
"completions/max_terminated_length": 339.0,
"completions/mean_length": 175.375,
"completions/mean_terminated_length": 175.375,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9792,
"grad_norm": 5.84846305847168,
"kl": 1.20361328125,
"learning_rate": 1e-06,
"loss": 0.0497,
"num_tokens": 16080534.0,
"reward": 4.77056884765625,
"reward_std": 2.0563995838165283,
"rewards/rm_reward_func/mean": 4.77056884765625,
"rewards/rm_reward_func/std": 4.059889316558838,
"step": 1224
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 481.0,
"completions/mean_length": 284.625,
"completions/mean_terminated_length": 269.4666748046875,
"completions/min_length": 62.0,
"completions/min_terminated_length": 62.0,
"epoch": 0.98,
"grad_norm": 8.25778865814209,
"kl": 2.14990234375,
"learning_rate": 1e-06,
"loss": 0.0486,
"num_tokens": 16091746.0,
"reward": 6.4002685546875,
"reward_std": 7.458786964416504,
"rewards/rm_reward_func/mean": 6.4002685546875,
"rewards/rm_reward_func/std": 13.408005714416504,
"step": 1225
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 341.875,
"completions/mean_terminated_length": 294.239990234375,
"completions/min_length": 131.0,
"completions/min_terminated_length": 131.0,
"epoch": 0.9808,
"grad_norm": 10.492647171020508,
"kl": 2.84521484375,
"learning_rate": 1e-06,
"loss": 0.0935,
"num_tokens": 16105958.0,
"reward": -4.466461181640625,
"reward_std": 5.502716064453125,
"rewards/rm_reward_func/mean": -4.466461181640625,
"rewards/rm_reward_func/std": 6.831468105316162,
"step": 1226
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.40625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 480.0,
"completions/mean_length": 310.53125,
"completions/mean_terminated_length": 172.68421936035156,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9816,
"grad_norm": 8.328527450561523,
"kl": 3.4453125,
"learning_rate": 1e-06,
"loss": 0.1165,
"num_tokens": 16120351.0,
"reward": -7.10888671875,
"reward_std": 3.2640366554260254,
"rewards/rm_reward_func/mean": -7.10888671875,
"rewards/rm_reward_func/std": 10.982377052307129,
"step": 1227
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 439.0,
"completions/max_terminated_length": 439.0,
"completions/mean_length": 246.4375,
"completions/mean_terminated_length": 246.4375,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9824,
"grad_norm": 10.678609848022461,
"kl": 3.5791015625,
"learning_rate": 1e-06,
"loss": 0.1131,
"num_tokens": 16133981.0,
"reward": 5.48870849609375,
"reward_std": 6.148322105407715,
"rewards/rm_reward_func/mean": 5.48870849609375,
"rewards/rm_reward_func/std": 13.595304489135742,
"step": 1228
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 470.0,
"completions/max_terminated_length": 470.0,
"completions/mean_length": 254.1875,
"completions/mean_terminated_length": 254.1875,
"completions/min_length": 97.0,
"completions/min_terminated_length": 97.0,
"epoch": 0.9832,
"grad_norm": 7.430377006530762,
"kl": 1.12890625,
"learning_rate": 1e-06,
"loss": -0.0441,
"num_tokens": 16145083.0,
"reward": 8.41162109375,
"reward_std": 8.556034088134766,
"rewards/rm_reward_func/mean": 8.41162109375,
"rewards/rm_reward_func/std": 13.60086727142334,
"step": 1229
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 477.0,
"completions/mean_length": 232.53125,
"completions/mean_terminated_length": 213.90000915527344,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.984,
"grad_norm": 5.466613292694092,
"kl": 1.316650390625,
"learning_rate": 1e-06,
"loss": -0.0194,
"num_tokens": 16157052.0,
"reward": 13.6875,
"reward_std": 7.069519996643066,
"rewards/rm_reward_func/mean": 13.6875,
"rewards/rm_reward_func/std": 15.511878967285156,
"step": 1230
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 479.0,
"completions/mean_length": 256.96875,
"completions/mean_terminated_length": 239.9666748046875,
"completions/min_length": 64.0,
"completions/min_terminated_length": 64.0,
"epoch": 0.9848,
"grad_norm": 14.960919380187988,
"kl": 1.33642578125,
"learning_rate": 1e-06,
"loss": 0.048,
"num_tokens": 16167787.0,
"reward": 13.341796875,
"reward_std": 7.339491844177246,
"rewards/rm_reward_func/mean": 13.341796875,
"rewards/rm_reward_func/std": 19.08413314819336,
"step": 1231
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 510.0,
"completions/mean_length": 271.5625,
"completions/mean_terminated_length": 263.80645751953125,
"completions/min_length": 102.0,
"completions/min_terminated_length": 102.0,
"epoch": 0.9856,
"grad_norm": 21.033065795898438,
"kl": 3.0126953125,
"learning_rate": 1e-06,
"loss": 0.0807,
"num_tokens": 16178821.0,
"reward": 1.5001983642578125,
"reward_std": 6.436107635498047,
"rewards/rm_reward_func/mean": 1.5001983642578125,
"rewards/rm_reward_func/std": 14.546778678894043,
"step": 1232
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 509.0,
"completions/mean_length": 342.78125,
"completions/mean_terminated_length": 286.375,
"completions/min_length": 73.0,
"completions/min_terminated_length": 73.0,
"epoch": 0.9864,
"grad_norm": 11.773747444152832,
"kl": 6.388427734375,
"learning_rate": 1e-06,
"loss": 0.3401,
"num_tokens": 16192638.0,
"reward": -4.092529296875,
"reward_std": 6.334899425506592,
"rewards/rm_reward_func/mean": -4.092529296875,
"rewards/rm_reward_func/std": 12.157096862792969,
"step": 1233
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.5,
"completions/max_length": 512.0,
"completions/max_terminated_length": 492.0,
"completions/mean_length": 448.21875,
"completions/mean_terminated_length": 384.4375,
"completions/min_length": 201.0,
"completions/min_terminated_length": 201.0,
"epoch": 0.9872,
"grad_norm": 14.910449028015137,
"kl": 4.833984375,
"learning_rate": 1e-06,
"loss": 0.1394,
"num_tokens": 16210957.0,
"reward": -4.65869140625,
"reward_std": 6.343834400177002,
"rewards/rm_reward_func/mean": -4.65869140625,
"rewards/rm_reward_func/std": 21.47957420349121,
"step": 1234
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.21875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 505.0,
"completions/mean_length": 258.75,
"completions/mean_terminated_length": 187.83999633789062,
"completions/min_length": 68.0,
"completions/min_terminated_length": 68.0,
"epoch": 0.988,
"grad_norm": 31.964218139648438,
"kl": 1.8095703125,
"learning_rate": 1e-06,
"loss": 0.0506,
"num_tokens": 16222165.0,
"reward": -6.220703125,
"reward_std": 4.152710914611816,
"rewards/rm_reward_func/mean": -6.220703125,
"rewards/rm_reward_func/std": 10.58847427368164,
"step": 1235
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 472.0,
"completions/max_terminated_length": 472.0,
"completions/mean_length": 198.71875,
"completions/mean_terminated_length": 198.71875,
"completions/min_length": 39.0,
"completions/min_terminated_length": 39.0,
"epoch": 0.9888,
"grad_norm": 5.49161958694458,
"kl": 1.7666015625,
"learning_rate": 1e-06,
"loss": -0.0291,
"num_tokens": 16231932.0,
"reward": -1.51971435546875,
"reward_std": 2.968614101409912,
"rewards/rm_reward_func/mean": -1.51971435546875,
"rewards/rm_reward_func/std": 9.40853500366211,
"step": 1236
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 511.0,
"completions/mean_length": 256.84375,
"completions/mean_terminated_length": 239.83334350585938,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9896,
"grad_norm": 7.274431228637695,
"kl": 0.58251953125,
"learning_rate": 1e-06,
"loss": -0.0121,
"num_tokens": 16243055.0,
"reward": 13.570831298828125,
"reward_std": 3.243131637573242,
"rewards/rm_reward_func/mean": 13.570831298828125,
"rewards/rm_reward_func/std": 12.231443405151367,
"step": 1237
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 508.0,
"completions/mean_length": 423.1875,
"completions/mean_terminated_length": 369.8999938964844,
"completions/min_length": 179.0,
"completions/min_terminated_length": 179.0,
"epoch": 0.9904,
"grad_norm": 8.700695037841797,
"kl": 1.9287109375,
"learning_rate": 1e-06,
"loss": 0.0465,
"num_tokens": 16259893.0,
"reward": -7.0559539794921875,
"reward_std": 5.43247652053833,
"rewards/rm_reward_func/mean": -7.0559539794921875,
"rewards/rm_reward_func/std": 7.467658519744873,
"step": 1238
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625,
"completions/max_length": 512.0,
"completions/max_terminated_length": 512.0,
"completions/mean_length": 320.25,
"completions/mean_terminated_length": 284.7407531738281,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9912,
"grad_norm": 14.817831039428711,
"kl": 1.71875,
"learning_rate": 1e-06,
"loss": 0.0421,
"num_tokens": 16273869.0,
"reward": 7.86328125,
"reward_std": 7.426191806793213,
"rewards/rm_reward_func/mean": 7.86328125,
"rewards/rm_reward_func/std": 9.81981372833252,
"step": 1239
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 500.0,
"completions/mean_length": 327.84375,
"completions/mean_terminated_length": 231.38095092773438,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.992,
"grad_norm": 10.064021110534668,
"kl": 1.5625,
"learning_rate": 1e-06,
"loss": 0.0699,
"num_tokens": 16288208.0,
"reward": 5.3447265625,
"reward_std": 4.9022955894470215,
"rewards/rm_reward_func/mean": 5.3447265625,
"rewards/rm_reward_func/std": 9.280166625976562,
"step": 1240
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 480.0,
"completions/max_terminated_length": 480.0,
"completions/mean_length": 348.71875,
"completions/mean_terminated_length": 348.71875,
"completions/min_length": 184.0,
"completions/min_terminated_length": 184.0,
"epoch": 0.9928,
"grad_norm": 24.340036392211914,
"kl": 0.641357421875,
"learning_rate": 1e-06,
"loss": 0.0044,
"num_tokens": 16301511.0,
"reward": 2.7781982421875,
"reward_std": 5.280330181121826,
"rewards/rm_reward_func/mean": 2.7781982421875,
"rewards/rm_reward_func/std": 7.673682689666748,
"step": 1241
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 499.0,
"completions/mean_length": 320.46875,
"completions/mean_terminated_length": 293.1071472167969,
"completions/min_length": 79.0,
"completions/min_terminated_length": 79.0,
"epoch": 0.9936,
"grad_norm": 5.585651397705078,
"kl": 2.2373046875,
"learning_rate": 1e-06,
"loss": 0.0234,
"num_tokens": 16314094.0,
"reward": 5.1142578125,
"reward_std": 4.644550323486328,
"rewards/rm_reward_func/mean": 5.1142578125,
"rewards/rm_reward_func/std": 6.202043056488037,
"step": 1242
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 483.0,
"completions/mean_length": 260.65625,
"completions/mean_terminated_length": 202.6538543701172,
"completions/min_length": 32.0,
"completions/min_terminated_length": 32.0,
"epoch": 0.9944,
"grad_norm": 38.17617416381836,
"kl": 7.8544921875,
"learning_rate": 1e-06,
"loss": 0.3338,
"num_tokens": 16327371.0,
"reward": -8.823577880859375,
"reward_std": 5.635340213775635,
"rewards/rm_reward_func/mean": -8.823577880859375,
"rewards/rm_reward_func/std": 15.1368989944458,
"step": 1243
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 302.96875,
"completions/mean_terminated_length": 281.3448181152344,
"completions/min_length": 55.0,
"completions/min_terminated_length": 55.0,
"epoch": 0.9952,
"grad_norm": 13.918490409851074,
"kl": 5.79296875,
"learning_rate": 1e-06,
"loss": 0.1535,
"num_tokens": 16340482.0,
"reward": -7.033935546875,
"reward_std": 8.026933670043945,
"rewards/rm_reward_func/mean": -7.033935546875,
"rewards/rm_reward_func/std": 13.346774101257324,
"step": 1244
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1875,
"completions/max_length": 512.0,
"completions/max_terminated_length": 495.0,
"completions/mean_length": 217.5625,
"completions/mean_terminated_length": 149.61538696289062,
"completions/min_length": 74.0,
"completions/min_terminated_length": 74.0,
"epoch": 0.996,
"grad_norm": 5.955099105834961,
"kl": 1.0595703125,
"learning_rate": 1e-06,
"loss": -0.0099,
"num_tokens": 16350628.0,
"reward": 6.19781494140625,
"reward_std": 5.490054130554199,
"rewards/rm_reward_func/mean": 6.19781494140625,
"rewards/rm_reward_func/std": 7.488785266876221,
"step": 1245
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09375,
"completions/max_length": 512.0,
"completions/max_terminated_length": 479.0,
"completions/mean_length": 313.625,
"completions/mean_terminated_length": 293.10345458984375,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9968,
"grad_norm": 16.472368240356445,
"kl": 6.9775390625,
"learning_rate": 1e-06,
"loss": 0.2136,
"num_tokens": 16364000.0,
"reward": -2.907135009765625,
"reward_std": 5.292792320251465,
"rewards/rm_reward_func/mean": -2.907135009765625,
"rewards/rm_reward_func/std": 9.868986129760742,
"step": 1246
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.03125,
"completions/max_length": 512.0,
"completions/max_terminated_length": 506.0,
"completions/mean_length": 282.6875,
"completions/mean_terminated_length": 275.2903137207031,
"completions/min_length": 50.0,
"completions/min_terminated_length": 50.0,
"epoch": 0.9976,
"grad_norm": 41.74302673339844,
"kl": 11.40576171875,
"learning_rate": 1e-06,
"loss": 0.2963,
"num_tokens": 16375222.0,
"reward": -6.39111328125,
"reward_std": 3.2877705097198486,
"rewards/rm_reward_func/mean": -6.39111328125,
"rewards/rm_reward_func/std": 19.594057083129883,
"step": 1247
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 512.0,
"completions/max_terminated_length": 473.0,
"completions/mean_length": 304.3125,
"completions/mean_terminated_length": 235.08334350585938,
"completions/min_length": 47.0,
"completions/min_terminated_length": 47.0,
"epoch": 0.9984,
"grad_norm": 23.52429962158203,
"kl": 5.68408203125,
"learning_rate": 1e-06,
"loss": -0.0276,
"num_tokens": 16387192.0,
"reward": -7.197265625,
"reward_std": 10.437725067138672,
"rewards/rm_reward_func/mean": -7.197265625,
"rewards/rm_reward_func/std": 14.508234024047852,
"step": 1248
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 441.0,
"completions/max_terminated_length": 441.0,
"completions/mean_length": 246.1875,
"completions/mean_terminated_length": 246.1875,
"completions/min_length": 80.0,
"completions/min_terminated_length": 80.0,
"epoch": 0.9992,
"grad_norm": 11.459190368652344,
"kl": 6.080078125,
"learning_rate": 1e-06,
"loss": 0.1161,
"num_tokens": 16397814.0,
"reward": -1.294921875,
"reward_std": 7.976162433624268,
"rewards/rm_reward_func/mean": -1.294921875,
"rewards/rm_reward_func/std": 13.71561336517334,
"step": 1249
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 485.0,
"completions/max_terminated_length": 485.0,
"completions/mean_length": 347.75,
"completions/mean_terminated_length": 347.75,
"completions/min_length": 136.0,
"completions/min_terminated_length": 136.0,
"epoch": 1.0,
"grad_norm": 22.80115509033203,
"kl": 4.734375,
"learning_rate": 1e-06,
"loss": 0.0654,
"num_tokens": 16412325.0,
"reward": -4.2939453125,
"reward_std": 7.761114120483398,
"rewards/rm_reward_func/mean": -4.2939453125,
"rewards/rm_reward_func/std": 18.258264541625977,
"step": 1250
}
],
"logging_steps": 1,
"max_steps": 1250,
"num_input_tokens_seen": 16412325,
"num_train_epochs": 1,
"save_steps": 250,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": true
},
"attributes": {}
}
},
"total_flos": 0.0,
"train_batch_size": 8,
"trial_name": null,
"trial_params": null
}