Files
csc415-phase1-0.5b-fast/checkpoint-2000/trainer_state.json

5635 lines
210 KiB
JSON
Raw Permalink Normal View History

{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 1.3080444735120995,
"eval_steps": 500,
"global_step": 2000,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.49000000953674316,
"completions/max_length": 384.0,
"completions/max_terminated_length": 372.8,
"completions/mean_length": 291.7250061035156,
"completions/mean_terminated_length": 204.1207077026367,
"completions/min_length": 15.7,
"completions/min_terminated_length": 15.7,
"entropy": 1.5640255227684974,
"epoch": 0.006540222367560497,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.5129233002662659,
"kl": 0.0005587212668615394,
"learning_rate": 4.985284499672989e-07,
"loss": -0.04552017748355865,
"num_tokens": 324400.0,
"reward": -2.107620294111712e-06,
"reward_std": 0.9998070120811462,
"rewards/rnd_reward/mean": -2.107620250058062e-06,
"rewards/rnd_reward/std": 0.9998070061206817,
"step": 10,
"step_time": 9.577712386706844
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.5312500178813935,
"completions/max_length": 384.0,
"completions/max_terminated_length": 374.2,
"completions/mean_length": 293.2887481689453,
"completions/mean_terminated_length": 190.75759887695312,
"completions/min_length": 18.8,
"completions/min_terminated_length": 18.8,
"entropy": 1.451838418841362,
"epoch": 0.013080444735120994,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.47466912865638733,
"kl": 0.000643831241177395,
"learning_rate": 4.968933943754087e-07,
"loss": 0.00032075121998786926,
"num_tokens": 647161.0,
"reward": -1.490712192975252e-06,
"reward_std": 0.9998168766498565,
"rewards/rnd_reward/mean": -1.4907121759222264e-06,
"rewards/rnd_reward/std": 0.9998168766498565,
"step": 20,
"step_time": 9.337269740481862
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.5450000017881393,
"completions/max_length": 384.0,
"completions/max_terminated_length": 368.6,
"completions/mean_length": 298.80250549316406,
"completions/mean_terminated_length": 196.99718322753907,
"completions/min_length": 20.3,
"completions/min_terminated_length": 20.3,
"entropy": 1.5065956979990005,
"epoch": 0.01962066710268149,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.5856828689575195,
"kl": 0.0006675069424090907,
"learning_rate": 4.952583387835186e-07,
"loss": 0.02223682850599289,
"num_tokens": 982698.0,
"reward": -1.2338162438396694e-07,
"reward_std": 0.9998676717281342,
"rewards/rnd_reward/mean": -1.233816414369926e-07,
"rewards/rnd_reward/std": 0.9998676896095275,
"step": 30,
"step_time": 9.97233030446805
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.5125000089406967,
"completions/max_length": 384.0,
"completions/max_terminated_length": 371.2,
"completions/mean_length": 287.2925048828125,
"completions/mean_terminated_length": 185.24396667480468,
"completions/min_length": 10.5,
"completions/min_terminated_length": 10.5,
"entropy": 1.6533239781856537,
"epoch": 0.026160889470241987,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.5665533542633057,
"kl": 0.0007774221347062848,
"learning_rate": 4.936232831916285e-07,
"loss": 0.06135546565055847,
"num_tokens": 1303862.0,
"reward": -6.012618625206301e-07,
"reward_std": 0.9998966813087463,
"rewards/rnd_reward/mean": -6.012618571915596e-07,
"rewards/rnd_reward/std": 0.9998966932296753,
"step": 40,
"step_time": 9.489675145596266
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.46375001072883604,
"completions/max_length": 384.0,
"completions/max_terminated_length": 377.3,
"completions/mean_length": 269.5550064086914,
"completions/mean_terminated_length": 170.72807769775392,
"completions/min_length": 7.2,
"completions/min_terminated_length": 7.2,
"entropy": 1.7108511567115783,
"epoch": 0.032701111837802485,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.7200465798377991,
"kl": 0.0023742157864035107,
"learning_rate": 4.919882275997384e-07,
"loss": 0.11103466749191285,
"num_tokens": 1618921.0,
"reward": -8.426606612488285e-08,
"reward_std": 0.9999245047569275,
"rewards/rnd_reward/mean": -8.426606257216918e-08,
"rewards/rnd_reward/std": 0.999924510717392,
"step": 50,
"step_time": 10.167311429185792
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.4012500047683716,
"completions/max_length": 384.0,
"completions/max_terminated_length": 365.1,
"completions/mean_length": 223.41500549316407,
"completions/mean_terminated_length": 115.63493576049805,
"completions/min_length": 2.8,
"completions/min_terminated_length": 2.8,
"entropy": 2.521006014943123,
"epoch": 0.03924133420536298,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.70113205909729,
"kl": 0.020581040543038397,
"learning_rate": 4.903531720078482e-07,
"loss": 0.17053321599960328,
"num_tokens": 1894883.0,
"reward": -1.1384487108578867e-07,
"reward_std": 0.9999224483966828,
"rewards/rnd_reward/mean": -1.1384487108578867e-07,
"rewards/rnd_reward/std": 0.9999224364757537,
"step": 60,
"step_time": 9.91588868314866
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3012500047683716,
"completions/max_length": 384.0,
"completions/max_terminated_length": 373.8,
"completions/mean_length": 183.6437515258789,
"completions/mean_terminated_length": 97.22235260009765,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.7396299123764036,
"epoch": 0.04578155657292348,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.7759609222412109,
"kl": 0.0463468745816499,
"learning_rate": 4.887181164159581e-07,
"loss": 0.14934378862380981,
"num_tokens": 2138653.0,
"reward": -4.6938660425155375e-08,
"reward_std": 0.9998835921287537,
"rewards/rnd_reward/mean": -4.6938658648798536e-08,
"rewards/rnd_reward/std": 0.9998836040496826,
"step": 70,
"step_time": 10.02495912197046
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.2825000062584877,
"completions/max_length": 384.0,
"completions/max_terminated_length": 360.8,
"completions/mean_length": 176.12750549316405,
"completions/mean_terminated_length": 94.6318603515625,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.7325789600610735,
"epoch": 0.052321778940483975,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8319016695022583,
"kl": 0.04594853101298213,
"learning_rate": 4.87083060824068e-07,
"loss": 0.018015310168266296,
"num_tokens": 2376420.0,
"reward": -4.187226378959963e-08,
"reward_std": 0.9998487293720245,
"rewards/rnd_reward/mean": -4.1872263079056894e-08,
"rewards/rnd_reward/std": 0.9998487293720245,
"step": 80,
"step_time": 9.701193272857926
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34750000238418577,
"completions/max_length": 384.0,
"completions/max_terminated_length": 363.4,
"completions/mean_length": 207.52375335693358,
"completions/mean_terminated_length": 114.01394119262696,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.342811721563339,
"epoch": 0.058862001308044476,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.5935761332511902,
"kl": 0.019986469263676553,
"learning_rate": 4.854480052321779e-07,
"loss": 0.010216768831014633,
"num_tokens": 2641029.0,
"reward": -2.0414591972439666e-08,
"reward_std": 0.9998277425765991,
"rewards/rnd_reward/mean": -2.0414591972439666e-08,
"rewards/rnd_reward/std": 0.9998277485370636,
"step": 90,
"step_time": 10.468820211617276
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.4150000065565109,
"completions/max_length": 384.0,
"completions/max_terminated_length": 356.7,
"completions/mean_length": 237.4625045776367,
"completions/mean_terminated_length": 134.2795196533203,
"completions/min_length": 3.0,
"completions/min_terminated_length": 3.0,
"entropy": 2.1418308436870577,
"epoch": 0.06540222367560497,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.5613327622413635,
"kl": 0.006808469316456467,
"learning_rate": 4.838129496402878e-07,
"loss": 0.04351484477519989,
"num_tokens": 2930639.0,
"reward": 8.493661918240747e-09,
"reward_std": 0.9997802138328552,
"rewards/rnd_reward/mean": 8.493660852426644e-09,
"rewards/rnd_reward/std": 0.9997801959514618,
"step": 100,
"step_time": 10.307341906265355
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.38875000178813934,
"completions/max_length": 384.0,
"completions/max_terminated_length": 370.8,
"completions/mean_length": 234.60625610351562,
"completions/mean_terminated_length": 139.61359405517578,
"completions/min_length": 2.4,
"completions/min_terminated_length": 2.4,
"entropy": 2.0046547770500185,
"epoch": 0.07194244604316546,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.7055774927139282,
"kl": 0.006759516580495983,
"learning_rate": 4.821778940483976e-07,
"loss": 0.0377700001001358,
"num_tokens": 3215169.0,
"reward": -2.920627686364696e-08,
"reward_std": 0.9997233927249909,
"rewards/rnd_reward/mean": -2.9206276153104228e-08,
"rewards/rnd_reward/std": 0.9997233927249909,
"step": 110,
"step_time": 9.7759924543323
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.38250001072883605,
"completions/max_length": 384.0,
"completions/max_terminated_length": 364.8,
"completions/mean_length": 221.19375457763672,
"completions/mean_terminated_length": 120.60059127807617,
"completions/min_length": 2.2,
"completions/min_terminated_length": 2.2,
"entropy": 2.0983517467975616,
"epoch": 0.07848266841072596,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.5727905631065369,
"kl": 0.009989026037510484,
"learning_rate": 4.805428384565075e-07,
"loss": 0.09626840949058532,
"num_tokens": 3496559.0,
"reward": 1.654028882747127e-08,
"reward_std": 0.9997243106365203,
"rewards/rnd_reward/mean": 1.654028882747127e-08,
"rewards/rnd_reward/std": 0.9997243225574494,
"step": 120,
"step_time": 9.981053173542023
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.34750000536441805,
"completions/max_length": 384.0,
"completions/max_terminated_length": 370.4,
"completions/mean_length": 207.67625274658204,
"completions/mean_terminated_length": 114.74080200195313,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.2568650126457213,
"epoch": 0.08502289077828647,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8493079543113708,
"kl": 0.014032764779403805,
"learning_rate": 4.789077828646174e-07,
"loss": -0.015352001786231995,
"num_tokens": 3759245.0,
"reward": -5.06639494801675e-09,
"reward_std": 0.9996613085269928,
"rewards/rnd_reward/mean": -5.066393526931279e-09,
"rewards/rnd_reward/std": 0.9996613144874573,
"step": 130,
"step_time": 10.069910727906972
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3137500062584877,
"completions/max_length": 384.0,
"completions/max_terminated_length": 373.4,
"completions/mean_length": 196.95000305175782,
"completions/mean_terminated_length": 111.34912643432617,
"completions/min_length": 2.2,
"completions/min_terminated_length": 2.2,
"entropy": 2.305853408575058,
"epoch": 0.09156311314584696,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.7982010245323181,
"kl": 0.016896341042593122,
"learning_rate": 4.772727272727273e-07,
"loss": 0.02038821578025818,
"num_tokens": 4010260.0,
"reward": -6.914138950264714e-08,
"reward_std": 0.9996093094348908,
"rewards/rnd_reward/mean": -6.914138808156168e-08,
"rewards/rnd_reward/std": 0.9996093094348908,
"step": 140,
"step_time": 9.98037786844652
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3700000047683716,
"completions/max_length": 384.0,
"completions/max_terminated_length": 355.3,
"completions/mean_length": 212.5687515258789,
"completions/mean_terminated_length": 111.88678207397462,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.2448606371879576,
"epoch": 0.09810333551340746,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.6012561321258545,
"kl": 0.01302192104049027,
"learning_rate": 4.756376716808371e-07,
"loss": 0.041692107915878296,
"num_tokens": 4270360.0,
"reward": -6.854534007061375e-09,
"reward_std": 0.9995533764362335,
"rewards/rnd_reward/mean": -6.854535428146846e-09,
"rewards/rnd_reward/std": 0.9995533883571625,
"step": 150,
"step_time": 9.479559903079643
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3687500059604645,
"completions/max_length": 384.0,
"completions/max_terminated_length": 373.0,
"completions/mean_length": 221.59500274658203,
"completions/mean_terminated_length": 127.51908645629882,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.2110244154930117,
"epoch": 0.10464355788096795,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.7059744000434875,
"kl": 0.009931368473917245,
"learning_rate": 4.74002616088947e-07,
"loss": 0.0420710027217865,
"num_tokens": 4538671.0,
"reward": -6.467104074481256e-08,
"reward_std": 0.9995002686977387,
"rewards/rnd_reward/mean": -6.467103932372709e-08,
"rewards/rnd_reward/std": 0.9995002567768096,
"step": 160,
"step_time": 9.463879713765346
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.38750000596046447,
"completions/max_length": 384.0,
"completions/max_terminated_length": 366.5,
"completions/mean_length": 224.09500274658203,
"completions/mean_terminated_length": 123.22678070068359,
"completions/min_length": 2.5,
"completions/min_terminated_length": 2.5,
"entropy": 2.105703830718994,
"epoch": 0.11118378024852844,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.6353176236152649,
"kl": 0.009572444634977728,
"learning_rate": 4.723675604970569e-07,
"loss": 0.057880669832229614,
"num_tokens": 4810167.0,
"reward": -1.3113021868349506e-08,
"reward_std": 0.9995246767997742,
"rewards/rnd_reward/mean": -1.3113020802535403e-08,
"rewards/rnd_reward/std": 0.9995246648788452,
"step": 170,
"step_time": 9.689974679402075
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3425000041723251,
"completions/max_length": 384.0,
"completions/max_terminated_length": 362.8,
"completions/mean_length": 207.19125213623047,
"completions/mean_terminated_length": 115.52461242675781,
"completions/min_length": 2.7,
"completions/min_terminated_length": 2.7,
"entropy": 2.3119001001119615,
"epoch": 0.11772400261608895,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.6543518900871277,
"kl": 0.01470447750762105,
"learning_rate": 4.7073250490516674e-07,
"loss": 0.05609138607978821,
"num_tokens": 5069705.0,
"reward": -3.21865089580875e-08,
"reward_std": 0.9993573248386383,
"rewards/rnd_reward/mean": -3.218650753700203e-08,
"rewards/rnd_reward/std": 0.9993573546409606,
"step": 180,
"step_time": 9.860351232904941
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.29625000804662704,
"completions/max_length": 384.0,
"completions/max_terminated_length": 361.4,
"completions/mean_length": 186.85875396728517,
"completions/mean_terminated_length": 103.74458618164063,
"completions/min_length": 2.2,
"completions/min_terminated_length": 2.2,
"entropy": 2.3917055457830427,
"epoch": 0.12426422498364945,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.6832460165023804,
"kl": 0.02189517344813794,
"learning_rate": 4.690974493132766e-07,
"loss": 0.0940218448638916,
"num_tokens": 5313337.0,
"reward": -2.801418297337932e-08,
"reward_std": 0.9993866741657257,
"rewards/rnd_reward/mean": -2.801418386155774e-08,
"rewards/rnd_reward/std": 0.9993866682052612,
"step": 190,
"step_time": 10.32580083089415
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.3087500035762787,
"completions/max_length": 384.0,
"completions/max_terminated_length": 362.0,
"completions/mean_length": 195.6250030517578,
"completions/mean_terminated_length": 111.06491088867188,
"completions/min_length": 2.2,
"completions/min_terminated_length": 2.2,
"entropy": 2.365246319770813,
"epoch": 0.13080444735120994,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.7736798524856567,
"kl": 0.018843610235489906,
"learning_rate": 4.674623937213865e-07,
"loss": 0.09085838794708252,
"num_tokens": 5564832.0,
"reward": -3.784895152847412e-08,
"reward_std": 0.9993082165718079,
"rewards/rnd_reward/mean": -3.7848948331031805e-08,
"rewards/rnd_reward/std": 0.9993082165718079,
"step": 200,
"step_time": 10.074973361729644
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.2800000086426735,
"completions/max_length": 384.0,
"completions/max_terminated_length": 364.2,
"completions/mean_length": 183.75125274658203,
"completions/mean_terminated_length": 105.59480438232421,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.3340355485677717,
"epoch": 0.13734466971877043,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.7310492992401123,
"kl": 0.019957996485754848,
"learning_rate": 4.6582733812949637e-07,
"loss": 0.09393368363380432,
"num_tokens": 5808593.0,
"reward": -7.003545956507651e-08,
"reward_std": 0.9992451012134552,
"rewards/rnd_reward/mean": -7.003545814399103e-08,
"rewards/rnd_reward/std": 0.9992450892925262,
"step": 210,
"step_time": 9.864037699718029
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.2600000038743019,
"completions/max_length": 384.0,
"completions/max_terminated_length": 367.5,
"completions/mean_length": 182.30500335693358,
"completions/mean_terminated_length": 111.45323791503907,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.3039004147052764,
"epoch": 0.14388489208633093,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.7140376567840576,
"kl": 0.020338369254022836,
"learning_rate": 4.641922825376063e-07,
"loss": 0.03314664363861084,
"num_tokens": 6039847.0,
"reward": -4.202127490415819e-08,
"reward_std": 0.9992108285427094,
"rewards/rnd_reward/mean": -4.202127419361546e-08,
"rewards/rnd_reward/std": 0.9992108345031738,
"step": 220,
"step_time": 9.367073794314638
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.2750000059604645,
"completions/max_length": 384.0,
"completions/max_terminated_length": 361.2,
"completions/mean_length": 176.20750274658204,
"completions/mean_terminated_length": 97.27564849853516,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.2686150133609773,
"epoch": 0.15042511445389142,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8464763760566711,
"kl": 0.021972918952815234,
"learning_rate": 4.6255722694571613e-07,
"loss": 0.04784287512302399,
"num_tokens": 6270078.0,
"reward": -1.1757016302738065e-07,
"reward_std": 0.9992310464382171,
"rewards/rnd_reward/mean": -1.1757016231683792e-07,
"rewards/rnd_reward/std": 0.9992310464382171,
"step": 230,
"step_time": 9.406287361658178
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.23750000298023224,
"completions/max_length": 384.0,
"completions/max_terminated_length": 358.0,
"completions/mean_length": 164.8237548828125,
"completions/mean_terminated_length": 96.10623779296876,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.4642169535160066,
"epoch": 0.15696533682145192,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.874134361743927,
"kl": 0.027495546033605933,
"learning_rate": 4.60922171353826e-07,
"loss": 0.09257912039756774,
"num_tokens": 6507582.0,
"reward": -2.2351741080228748e-08,
"reward_std": 0.9989596605300903,
"rewards/rnd_reward/mean": -2.2351743211856957e-08,
"rewards/rnd_reward/std": 0.9989596605300903,
"step": 240,
"step_time": 10.88162840725854
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.26375000476837157,
"completions/max_length": 384.0,
"completions/max_terminated_length": 368.7,
"completions/mean_length": 179.21375274658203,
"completions/mean_terminated_length": 105.97551345825195,
"completions/min_length": 2.2,
"completions/min_terminated_length": 2.2,
"entropy": 2.3090430706739427,
"epoch": 0.16350555918901244,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.7480731010437012,
"kl": 0.022477348684333265,
"learning_rate": 4.592871157619359e-07,
"loss": 0.0661897599697113,
"num_tokens": 6746233.0,
"reward": -8.285045609568442e-08,
"reward_std": 0.9989501237869263,
"rewards/rnd_reward/mean": -8.285045680622715e-08,
"rewards/rnd_reward/std": 0.9989501178264618,
"step": 250,
"step_time": 9.837229801248759
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.23000000417232513,
"completions/max_length": 384.0,
"completions/max_terminated_length": 365.9,
"completions/mean_length": 166.79125366210937,
"completions/mean_terminated_length": 101.81034469604492,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.3430103093385695,
"epoch": 0.17004578155657293,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.7986056804656982,
"kl": 0.026151024643331765,
"learning_rate": 4.5765206017004576e-07,
"loss": 0.1205091118812561,
"num_tokens": 6976091.0,
"reward": -7.808208817294826e-08,
"reward_std": 0.9988620162010193,
"rewards/rnd_reward/mean": -7.808208604132006e-08,
"rewards/rnd_reward/std": 0.9988620281219482,
"step": 260,
"step_time": 9.811681813630276
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.23125000298023224,
"completions/max_length": 384.0,
"completions/max_terminated_length": 360.4,
"completions/mean_length": 162.19500274658202,
"completions/mean_terminated_length": 95.31229019165039,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.2510120779275895,
"epoch": 0.17658600392413343,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.6842794418334961,
"kl": 0.025852510519325734,
"learning_rate": 4.5601700457815564e-07,
"loss": 0.07729284763336182,
"num_tokens": 7199527.0,
"reward": -7.122755136812487e-08,
"reward_std": 0.9989409029483796,
"rewards/rnd_reward/mean": -7.122754994703939e-08,
"rewards/rnd_reward/std": 0.9989409029483796,
"step": 270,
"step_time": 9.695072391931898
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.24750000536441802,
"completions/max_length": 384.0,
"completions/max_terminated_length": 347.6,
"completions/mean_length": 165.32125244140624,
"completions/mean_terminated_length": 93.36570816040039,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.2813161820173264,
"epoch": 0.18312622629169392,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8117495775222778,
"kl": 0.02381342255976051,
"learning_rate": 4.543819489862655e-07,
"loss": 0.08756870031356812,
"num_tokens": 7423314.0,
"reward": -3.010034372863402e-08,
"reward_std": 0.9987568318843841,
"rewards/rnd_reward/mean": -3.0100344261541065e-08,
"rewards/rnd_reward/std": 0.9987568318843841,
"step": 280,
"step_time": 9.538212020602078
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.27250000536441804,
"completions/max_length": 384.0,
"completions/max_terminated_length": 351.5,
"completions/mean_length": 171.89125366210936,
"completions/mean_terminated_length": 91.7330307006836,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.2805801182985306,
"epoch": 0.18966644865925442,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.7356550693511963,
"kl": 0.021292086993344127,
"learning_rate": 4.527468933943754e-07,
"loss": 0.0787746787071228,
"num_tokens": 7649017.0,
"reward": -2.2798775312082853e-08,
"reward_std": 0.9987796723842621,
"rewards/rnd_reward/mean": -2.279877513444717e-08,
"rewards/rnd_reward/std": 0.99877969622612,
"step": 290,
"step_time": 9.4448504867265
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 384.0,
"completions/max_terminated_length": 368.4,
"completions/mean_length": 172.06375274658203,
"completions/mean_terminated_length": 101.12049560546875,
"completions/min_length": 2.2,
"completions/min_terminated_length": 2.2,
"entropy": 2.062102970480919,
"epoch": 0.1962066710268149,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.678175151348114,
"kl": 0.021230081818066536,
"learning_rate": 4.511118378024852e-07,
"loss": 0.10422745943069459,
"num_tokens": 7893248.0,
"reward": 1.7285347131412722e-08,
"reward_std": 0.9987858593463897,
"rewards/rnd_reward/mean": 1.7285347841955455e-08,
"rewards/rnd_reward/std": 0.9987858593463897,
"step": 300,
"step_time": 11.542911725770683
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.23125,
"completions/max_length": 384.0,
"completions/max_terminated_length": 358.3,
"completions/mean_length": 155.94375305175782,
"completions/mean_terminated_length": 87.63837699890136,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.110625982284546,
"epoch": 0.2027468933943754,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8758037090301514,
"kl": 0.025008385512046517,
"learning_rate": 4.4947678221059515e-07,
"loss": 0.0913781225681305,
"num_tokens": 8105838.0,
"reward": -6.72042411054008e-08,
"reward_std": 0.9986489951610565,
"rewards/rnd_reward/mean": -6.720423826322985e-08,
"rewards/rnd_reward/std": 0.9986490130424499,
"step": 310,
"step_time": 9.397380503430032
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.19500000327825545,
"completions/max_length": 384.0,
"completions/max_terminated_length": 360.1,
"completions/mean_length": 141.38000259399413,
"completions/mean_terminated_length": 82.59085083007812,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.311973440647125,
"epoch": 0.2092871157619359,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9461565613746643,
"kl": 0.03633355665951967,
"learning_rate": 4.4784172661870503e-07,
"loss": 0.08147850632667542,
"num_tokens": 8325122.0,
"reward": -9.238719300697084e-09,
"reward_std": 0.9986236810684204,
"rewards/rnd_reward/mean": -9.238719300697084e-09,
"rewards/rnd_reward/std": 0.9986236929893494,
"step": 320,
"step_time": 11.082472172076814
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.20000000447034835,
"completions/max_length": 384.0,
"completions/max_terminated_length": 361.7,
"completions/mean_length": 141.80000228881835,
"completions/mean_terminated_length": 81.46792144775391,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.3696460843086244,
"epoch": 0.2158273381294964,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8129741549491882,
"kl": 0.034535441920161246,
"learning_rate": 4.462066710268149e-07,
"loss": 0.06801753044128418,
"num_tokens": 8535267.0,
"reward": -8.24779284869237e-08,
"reward_std": 0.9986066579818725,
"rewards/rnd_reward/mean": -8.24779284869237e-08,
"rewards/rnd_reward/std": 0.9986066579818725,
"step": 330,
"step_time": 9.98545232352335
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.2100000038743019,
"completions/max_length": 384.0,
"completions/max_terminated_length": 364.4,
"completions/mean_length": 141.52500228881837,
"completions/mean_terminated_length": 77.51832160949706,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.3277006804943086,
"epoch": 0.2223675604970569,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8912848234176636,
"kl": 0.0404030725825578,
"learning_rate": 4.445716154349248e-07,
"loss": 0.09486417174339294,
"num_tokens": 8741817.0,
"reward": -1.6838312300038183e-08,
"reward_std": 0.9984602987766266,
"rewards/rnd_reward/mean": -1.683831301058092e-08,
"rewards/rnd_reward/std": 0.9984602928161621,
"step": 340,
"step_time": 9.806188964121976
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1912500038743019,
"completions/max_length": 384.0,
"completions/max_terminated_length": 358.6,
"completions/mean_length": 139.37500076293946,
"completions/mean_terminated_length": 81.56527061462403,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.3621277123689652,
"epoch": 0.2289077828646174,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9253888130187988,
"kl": 0.03642263817600906,
"learning_rate": 4.429365598430346e-07,
"loss": 0.09143019318580628,
"num_tokens": 8945182.0,
"reward": -2.4735924242236252e-08,
"reward_std": 0.9984571993350982,
"rewards/rnd_reward/mean": -2.4735929926578137e-08,
"rewards/rnd_reward/std": 0.9984571933746338,
"step": 350,
"step_time": 9.939818266802467
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.18750000447034837,
"completions/max_length": 384.0,
"completions/max_terminated_length": 371.7,
"completions/mean_length": 138.01250076293945,
"completions/mean_terminated_length": 81.37003707885742,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.270884335041046,
"epoch": 0.2354480052321779,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8366574645042419,
"kl": 0.03530883879866451,
"learning_rate": 4.4130150425114454e-07,
"loss": 0.10644701719284058,
"num_tokens": 9147587.0,
"reward": -1.15185978266652e-07,
"reward_std": 0.9983633041381836,
"rewards/rnd_reward/mean": -1.1518597684556652e-07,
"rewards/rnd_reward/std": 0.9983633100986481,
"step": 360,
"step_time": 9.722514500608668
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.17000000029802323,
"completions/max_length": 384.0,
"completions/max_terminated_length": 366.5,
"completions/mean_length": 130.59125213623048,
"completions/mean_terminated_length": 79.17214775085449,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.3677036464214325,
"epoch": 0.2419882275997384,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.964614987373352,
"kl": 0.043067421624436976,
"learning_rate": 4.3966644865925437e-07,
"loss": 0.08349081873893738,
"num_tokens": 9340010.0,
"reward": -3.0696392450124675e-08,
"reward_std": 0.9982355415821076,
"rewards/rnd_reward/mean": -3.0696389607953736e-08,
"rewards/rnd_reward/std": 0.9982355415821076,
"step": 370,
"step_time": 9.710164245427586
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1887500010430813,
"completions/max_length": 384.0,
"completions/max_terminated_length": 367.2,
"completions/mean_length": 138.41875305175782,
"completions/mean_terminated_length": 81.47324066162109,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.258895403146744,
"epoch": 0.2485284499672989,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.7839703559875488,
"kl": 0.039375259960070255,
"learning_rate": 4.3803139306736425e-07,
"loss": 0.08618925213813781,
"num_tokens": 9545205.0,
"reward": -1.0460615040841503e-07,
"reward_std": 0.998303747177124,
"rewards/rnd_reward/mean": -1.046061518295005e-07,
"rewards/rnd_reward/std": 0.9983037531375885,
"step": 380,
"step_time": 9.776164471404627
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1625000037252903,
"completions/max_length": 384.0,
"completions/max_terminated_length": 342.9,
"completions/mean_length": 123.14125137329101,
"completions/mean_terminated_length": 72.26187210083008,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.2043879717588424,
"epoch": 0.25506867233485936,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.013533353805542,
"kl": 0.045199727965518834,
"learning_rate": 4.363963374754742e-07,
"loss": 0.08801217675209046,
"num_tokens": 9732323.0,
"reward": -1.3418495639072603e-07,
"reward_std": 0.9981010854244232,
"rewards/rnd_reward/mean": -1.3418495496964057e-07,
"rewards/rnd_reward/std": 0.9981010735034943,
"step": 390,
"step_time": 9.470399480173365
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.16500000208616256,
"completions/max_length": 384.0,
"completions/max_terminated_length": 357.9,
"completions/mean_length": 126.54125213623047,
"completions/mean_terminated_length": 75.62271652221679,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.1946812689304354,
"epoch": 0.2616088947024199,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9289289712905884,
"kl": 0.043896416388452056,
"learning_rate": 4.34761281883584e-07,
"loss": 0.06013514399528504,
"num_tokens": 9923251.0,
"reward": -1.1950731622079046e-07,
"reward_std": 0.9980400562286377,
"rewards/rnd_reward/mean": -1.19507314799705e-07,
"rewards/rnd_reward/std": 0.9980400741100312,
"step": 400,
"step_time": 9.578276086249389
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.19000000283122062,
"completions/max_length": 384.0,
"completions/max_terminated_length": 361.4,
"completions/mean_length": 142.2825019836426,
"completions/mean_terminated_length": 85.19309997558594,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.3042476952075956,
"epoch": 0.2681491170699804,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8114489316940308,
"kl": 0.039571146154776214,
"learning_rate": 4.331262262916939e-07,
"loss": 0.08236328959465027,
"num_tokens": 10132682.0,
"reward": -1.4752151500374566e-08,
"reward_std": 0.9982209146022797,
"rewards/rnd_reward/mean": -1.4752152210917303e-08,
"rewards/rnd_reward/std": 0.9982209086418152,
"step": 410,
"step_time": 9.807740885205567
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.20625000149011613,
"completions/max_length": 384.0,
"completions/max_terminated_length": 351.8,
"completions/mean_length": 149.2062530517578,
"completions/mean_terminated_length": 88.13671951293945,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.38392159640789,
"epoch": 0.27468933943754087,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.7539530396461487,
"kl": 0.03540583737194538,
"learning_rate": 4.3149117069980376e-07,
"loss": 0.08869684934616089,
"num_tokens": 10345522.0,
"reward": -1.3530254712179611e-07,
"reward_std": 0.9981609523296356,
"rewards/rnd_reward/mean": -1.3530254356908243e-07,
"rewards/rnd_reward/std": 0.9981609761714936,
"step": 420,
"step_time": 9.619695667061023
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.18750000223517418,
"completions/max_length": 384.0,
"completions/max_terminated_length": 357.4,
"completions/mean_length": 139.9700035095215,
"completions/mean_terminated_length": 83.52344436645508,
"completions/min_length": 2.2,
"completions/min_terminated_length": 2.2,
"entropy": 2.225811779499054,
"epoch": 0.2812295618051014,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.7882169485092163,
"kl": 0.031479011243209246,
"learning_rate": 4.2985611510791364e-07,
"loss": 0.09766495227813721,
"num_tokens": 10563093.0,
"reward": -1.4021993024471158e-07,
"reward_std": 0.998052579164505,
"rewards/rnd_reward/mean": -1.4021992598145516e-07,
"rewards/rnd_reward/std": 0.9980525553226471,
"step": 430,
"step_time": 11.129851599340327
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.18875000178813933,
"completions/max_length": 384.0,
"completions/max_terminated_length": 360.8,
"completions/mean_length": 143.57625198364258,
"completions/mean_terminated_length": 87.31694107055664,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.1123261958360673,
"epoch": 0.28776978417266186,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.7737316489219666,
"kl": 0.0280194781254977,
"learning_rate": 4.2822105951602357e-07,
"loss": 0.10439527034759521,
"num_tokens": 10766159.0,
"reward": -4.2766332430943524e-08,
"reward_std": 0.998040771484375,
"rewards/rnd_reward/mean": -4.276633118749373e-08,
"rewards/rnd_reward/std": 0.998040747642517,
"step": 440,
"step_time": 9.481722956942395
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.16625000312924385,
"completions/max_length": 384.0,
"completions/max_terminated_length": 365.1,
"completions/mean_length": 131.25125427246093,
"completions/mean_terminated_length": 80.89927139282227,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.2791686803102493,
"epoch": 0.2943100065402224,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.834428608417511,
"kl": 0.04166741594672203,
"learning_rate": 4.265860039241334e-07,
"loss": 0.07646016478538513,
"num_tokens": 10965315.0,
"reward": -1.0013580258316779e-07,
"reward_std": 0.998044753074646,
"rewards/rnd_reward/mean": -1.0013580116208232e-07,
"rewards/rnd_reward/std": 0.9980447709560394,
"step": 450,
"step_time": 10.756261567212642
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.13750000298023224,
"completions/max_length": 384.0,
"completions/max_terminated_length": 361.9,
"completions/mean_length": 118.16625137329102,
"completions/mean_terminated_length": 75.56828384399414,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.3162699341773987,
"epoch": 0.30085022890778285,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.1823770999908447,
"kl": 0.04914104142226279,
"learning_rate": 4.249509483322433e-07,
"loss": 0.09161773324012756,
"num_tokens": 11151658.0,
"reward": 1.4454126073815132e-08,
"reward_std": 0.9979428768157959,
"rewards/rnd_reward/mean": 1.4454126073815132e-08,
"rewards/rnd_reward/std": 0.9979428887367249,
"step": 460,
"step_time": 9.690541778854094
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15625000596046448,
"completions/max_length": 384.0,
"completions/max_terminated_length": 359.0,
"completions/mean_length": 129.17875442504882,
"completions/mean_terminated_length": 81.90925521850586,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.250484657287598,
"epoch": 0.30739045127534337,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.7849432229995728,
"kl": 0.044866825267672536,
"learning_rate": 4.2331589274035315e-07,
"loss": 0.1366190195083618,
"num_tokens": 11350266.0,
"reward": -5.528331037396583e-08,
"reward_std": 0.997910988330841,
"rewards/rnd_reward/mean": -5.528330859760899e-08,
"rewards/rnd_reward/std": 0.997911012172699,
"step": 470,
"step_time": 10.094013441796415
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.14375000149011613,
"completions/max_length": 384.0,
"completions/max_terminated_length": 355.1,
"completions/mean_length": 121.7000015258789,
"completions/mean_terminated_length": 77.64777450561523,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.356913906335831,
"epoch": 0.31393067364290383,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.7464061975479126,
"kl": 0.047594004729762676,
"learning_rate": 4.2168083714846303e-07,
"loss": 0.093714839220047,
"num_tokens": 11546191.0,
"reward": -2.1904707381281696e-08,
"reward_std": 0.9980335652828216,
"rewards/rnd_reward/mean": -2.1904707381281696e-08,
"rewards/rnd_reward/std": 0.9980335593223572,
"step": 480,
"step_time": 10.49183724487666
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1312500022351742,
"completions/max_length": 384.0,
"completions/max_terminated_length": 357.0,
"completions/mean_length": 110.71375274658203,
"completions/mean_terminated_length": 69.09722747802735,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.3989109843969345,
"epoch": 0.32047089601046436,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9596973061561584,
"kl": 0.059090607427060605,
"learning_rate": 4.200457815565729e-07,
"loss": 0.061995261907577516,
"num_tokens": 11729672.0,
"reward": -8.441507906020718e-08,
"reward_std": 0.9977342307567596,
"rewards/rnd_reward/mean": -8.441507870493581e-08,
"rewards/rnd_reward/std": 0.9977342247962951,
"step": 490,
"step_time": 9.824798973021098
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.13000000193715094,
"completions/max_length": 384.0,
"completions/max_terminated_length": 361.2,
"completions/mean_length": 110.2125015258789,
"completions/mean_terminated_length": 69.26982688903809,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.335902637243271,
"epoch": 0.3270111183780249,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.0347245931625366,
"kl": 0.06009262367151678,
"learning_rate": 4.184107259646828e-07,
"loss": 0.011750234663486481,
"num_tokens": 11908552.0,
"reward": -7.688999374977356e-08,
"reward_std": 0.9977168381214142,
"rewards/rnd_reward/mean": -7.688999019705988e-08,
"rewards/rnd_reward/std": 0.9977168321609498,
"step": 500,
"step_time": 9.731353901792318
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1450000025331974,
"completions/max_length": 384.0,
"completions/max_terminated_length": 360.2,
"completions/mean_length": 118.08375244140625,
"completions/mean_terminated_length": 73.00995483398438,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.557803976535797,
"epoch": 0.33355134074558535,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9074205160140991,
"kl": 0.06632269411347806,
"learning_rate": 4.1677567037279267e-07,
"loss": 0.08373542428016663,
"num_tokens": 12103254.0,
"reward": -1.1697411828048843e-07,
"reward_std": 0.9978465795516968,
"rewards/rnd_reward/mean": -1.1697411828048843e-07,
"rewards/rnd_reward/std": 0.9978465795516968,
"step": 510,
"step_time": 10.207280423282645
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15875000208616258,
"completions/max_length": 384.0,
"completions/max_terminated_length": 361.2,
"completions/mean_length": 123.88500137329102,
"completions/mean_terminated_length": 74.63431396484376,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.437187620997429,
"epoch": 0.34009156311314587,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8857874870300293,
"kl": 0.06048318287357688,
"learning_rate": 4.151406147809025e-07,
"loss": 0.037069684267044066,
"num_tokens": 12302092.0,
"reward": -3.874301874873254e-08,
"reward_std": 0.9979513227939606,
"rewards/rnd_reward/mean": -3.874301874873254e-08,
"rewards/rnd_reward/std": 0.9979513168334961,
"step": 520,
"step_time": 10.34260532106273
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.17000000253319741,
"completions/max_length": 384.0,
"completions/max_terminated_length": 358.1,
"completions/mean_length": 132.52125396728516,
"completions/mean_terminated_length": 80.95349197387695,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.4997961223125458,
"epoch": 0.34663178548070633,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9028640985488892,
"kl": 0.05214450396597385,
"learning_rate": 4.135055591890124e-07,
"loss": 0.07458721995353698,
"num_tokens": 12499434.0,
"reward": -2.2247433655309124e-07,
"reward_std": 0.9978251278400421,
"rewards/rnd_reward/mean": -2.2247433655309124e-07,
"rewards/rnd_reward/std": 0.9978251218795776,
"step": 530,
"step_time": 9.664499558508396
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15125000178813935,
"completions/max_length": 384.0,
"completions/max_terminated_length": 354.4,
"completions/mean_length": 120.7087516784668,
"completions/mean_terminated_length": 73.79461898803712,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.547806602716446,
"epoch": 0.35317200784826686,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.0128388404846191,
"kl": 0.05931666204705834,
"learning_rate": 4.118705035971223e-07,
"loss": 0.11692253351211548,
"num_tokens": 12683096.0,
"reward": -8.344649238267721e-09,
"reward_std": 0.9977494418621063,
"rewards/rnd_reward/mean": -8.344651369895927e-09,
"rewards/rnd_reward/std": 0.9977494418621063,
"step": 540,
"step_time": 9.593502512201667
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.13500000387430192,
"completions/max_length": 384.0,
"completions/max_terminated_length": 349.6,
"completions/mean_length": 111.18750305175782,
"completions/mean_terminated_length": 68.47658958435059,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.681893822550774,
"epoch": 0.3597122302158273,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.1282111406326294,
"kl": 0.06902076229453087,
"learning_rate": 4.1023544800523213e-07,
"loss": 0.023469969630241394,
"num_tokens": 12862906.0,
"reward": -5.677342400645102e-08,
"reward_std": 0.9978136360645294,
"rewards/rnd_reward/mean": -5.6773422585365554e-08,
"rewards/rnd_reward/std": 0.997813630104065,
"step": 550,
"step_time": 9.778036265238189
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.12000000327825547,
"completions/max_length": 384.0,
"completions/max_terminated_length": 358.0,
"completions/mean_length": 104.53875045776367,
"completions/mean_terminated_length": 66.3183536529541,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.5951172381639482,
"epoch": 0.36625245258338784,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.1439512968063354,
"kl": 0.07057485273107886,
"learning_rate": 4.0860039241334206e-07,
"loss": 0.1278893232345581,
"num_tokens": 13049132.0,
"reward": -1.1786818543413347e-07,
"reward_std": 0.9977734804153442,
"rewards/rnd_reward/mean": -1.17868184013048e-07,
"rewards/rnd_reward/std": 0.9977734982967377,
"step": 560,
"step_time": 10.389736549765804
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.10500000044703484,
"completions/max_length": 384.0,
"completions/max_terminated_length": 346.3,
"completions/mean_length": 99.66875076293945,
"completions/mean_terminated_length": 66.29649963378907,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.6503134578466416,
"epoch": 0.3727926749509483,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.0748413801193237,
"kl": 0.08892364026978612,
"learning_rate": 4.069653368214519e-07,
"loss": 0.13109040260314941,
"num_tokens": 13221157.0,
"reward": -1.4342367933295462e-07,
"reward_std": 0.997740238904953,
"rewards/rnd_reward/mean": -1.4342367791186916e-07,
"rewards/rnd_reward/std": 0.9977402448654175,
"step": 570,
"step_time": 9.71618521134369
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.12250000312924385,
"completions/max_length": 384.0,
"completions/max_terminated_length": 351.8,
"completions/mean_length": 110.21125030517578,
"completions/mean_terminated_length": 71.826904296875,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.702208098769188,
"epoch": 0.37933289731850883,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.0691598653793335,
"kl": 0.07306001400575042,
"learning_rate": 4.053302812295618e-07,
"loss": 0.09580587744712829,
"num_tokens": 13409746.0,
"reward": -1.6450882269758438e-07,
"reward_std": 0.9975189864635468,
"rewards/rnd_reward/mean": -1.6450882198704164e-07,
"rewards/rnd_reward/std": 0.9975189983844757,
"step": 580,
"step_time": 10.602098648622633
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.13250000216066837,
"completions/max_length": 384.0,
"completions/max_terminated_length": 348.2,
"completions/mean_length": 118.48125228881835,
"completions/mean_terminated_length": 77.98408432006836,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.5411072075366974,
"epoch": 0.3858731196860693,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8738215565681458,
"kl": 0.06685408288612962,
"learning_rate": 4.0369522563767164e-07,
"loss": 0.0531582236289978,
"num_tokens": 13603691.0,
"reward": -8.791685381925874e-08,
"reward_std": 0.9976041257381439,
"rewards/rnd_reward/mean": -8.791684955600231e-08,
"rewards/rnd_reward/std": 0.9976041257381439,
"step": 590,
"step_time": 10.606248078867793
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.13375000208616256,
"completions/max_length": 384.0,
"completions/max_terminated_length": 356.1,
"completions/mean_length": 120.9775001525879,
"completions/mean_terminated_length": 80.35308418273925,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.454717567563057,
"epoch": 0.3924133420536298,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8597258925437927,
"kl": 0.06251584179699421,
"learning_rate": 4.020601700457815e-07,
"loss": 0.10086486339569092,
"num_tokens": 13793088.0,
"reward": -4.872679788547885e-08,
"reward_std": 0.997677344083786,
"rewards/rnd_reward/mean": -4.872679859602158e-08,
"rewards/rnd_reward/std": 0.997677332162857,
"step": 600,
"step_time": 9.838808755297213
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.13875000327825546,
"completions/max_length": 384.0,
"completions/max_terminated_length": 340.2,
"completions/mean_length": 112.62250213623047,
"completions/mean_terminated_length": 69.04776878356934,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.5064212560653685,
"epoch": 0.39895356442119034,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.0397368669509888,
"kl": 0.06775042302906513,
"learning_rate": 4.0042511445389145e-07,
"loss": 0.14560239315032958,
"num_tokens": 13975906.0,
"reward": -1.70022253698221e-07,
"reward_std": 0.997452187538147,
"rewards/rnd_reward/mean": -1.7002224872442184e-07,
"rewards/rnd_reward/std": 0.9974521696567535,
"step": 610,
"step_time": 9.711915594828316
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.11000000387430191,
"completions/max_length": 384.0,
"completions/max_terminated_length": 360.5,
"completions/mean_length": 105.6062515258789,
"completions/mean_terminated_length": 71.07123832702636,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.531564822793007,
"epoch": 0.4054937867887508,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8839622139930725,
"kl": 0.07682965183630586,
"learning_rate": 3.987900588620013e-07,
"loss": 0.06944795846939086,
"num_tokens": 14154656.0,
"reward": 5.662443069986978e-09,
"reward_std": 0.9975684463977814,
"rewards/rnd_reward/mean": 5.662442181808558e-09,
"rewards/rnd_reward/std": 0.9975684404373169,
"step": 620,
"step_time": 10.306970553193242
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.10375000201165677,
"completions/max_length": 384.0,
"completions/max_terminated_length": 359.6,
"completions/mean_length": 106.93500137329102,
"completions/mean_terminated_length": 74.97318267822266,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.859010672569275,
"epoch": 0.41203400915631133,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.103442907333374,
"kl": 0.08963548736646772,
"learning_rate": 3.9715500327011115e-07,
"loss": 0.033435261249542235,
"num_tokens": 14327599.0,
"reward": -1.2174248684715394e-07,
"reward_std": 0.9974590480327606,
"rewards/rnd_reward/mean": -1.2174248613661122e-07,
"rewards/rnd_reward/std": 0.9974590301513672,
"step": 630,
"step_time": 9.607878714823164
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.12625000327825547,
"completions/max_length": 384.0,
"completions/max_terminated_length": 343.2,
"completions/mean_length": 108.95125045776368,
"completions/mean_terminated_length": 69.31592407226563,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.823798879981041,
"epoch": 0.4185742315238718,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.1582435369491577,
"kl": 0.09182370994240045,
"learning_rate": 3.9551994767822103e-07,
"loss": 0.06608177423477173,
"num_tokens": 14513650.0,
"reward": -1.3239681819765537e-07,
"reward_std": 0.9976047217845917,
"rewards/rnd_reward/mean": -1.3239681961874084e-07,
"rewards/rnd_reward/std": 0.9976047277450562,
"step": 640,
"step_time": 10.307288937154226
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.11000000163912774,
"completions/max_length": 384.0,
"completions/max_terminated_length": 361.8,
"completions/mean_length": 107.63500213623047,
"completions/mean_terminated_length": 73.39098167419434,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.695296034216881,
"epoch": 0.4251144538914323,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.0280959606170654,
"kl": 0.09326308779418468,
"learning_rate": 3.938848920863309e-07,
"loss": 0.030906683206558226,
"num_tokens": 14693328.0,
"reward": -1.2964010291938167e-07,
"reward_std": 0.9975219249725342,
"rewards/rnd_reward/mean": -1.2964010327465302e-07,
"rewards/rnd_reward/std": 0.9975219309329987,
"step": 650,
"step_time": 10.01973620983772
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1512500040233135,
"completions/max_length": 384.0,
"completions/max_terminated_length": 352.3,
"completions/mean_length": 129.62250213623048,
"completions/mean_terminated_length": 84.16030578613281,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.613438582420349,
"epoch": 0.4316546762589928,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9445599317550659,
"kl": 0.0699499910697341,
"learning_rate": 3.922498364944408e-07,
"loss": 0.08974806666374206,
"num_tokens": 14892686.0,
"reward": -4.194676717084178e-08,
"reward_std": 0.9975471794605255,
"rewards/rnd_reward/mean": -4.1946770057421645e-08,
"rewards/rnd_reward/std": 0.99754718542099,
"step": 660,
"step_time": 10.161354359728284
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.14750000312924386,
"completions/max_length": 384.0,
"completions/max_terminated_length": 357.7,
"completions/mean_length": 126.59500045776367,
"completions/mean_terminated_length": 81.99441680908203,
"completions/min_length": 2.3,
"completions/min_terminated_length": 2.3,
"entropy": 2.7322494119405745,
"epoch": 0.4381948986265533,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9810017943382263,
"kl": 0.08053934266790748,
"learning_rate": 3.9061478090255067e-07,
"loss": 0.14365124702453613,
"num_tokens": 15083047.0,
"reward": -7.979571803318208e-08,
"reward_std": 0.9972860395908356,
"rewards/rnd_reward/mean": -7.979571803318208e-08,
"rewards/rnd_reward/std": 0.9972860395908356,
"step": 670,
"step_time": 9.683341883006506
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.14250000119209288,
"completions/max_length": 384.0,
"completions/max_terminated_length": 351.5,
"completions/mean_length": 125.95000381469727,
"completions/mean_terminated_length": 83.18684883117676,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.6392508417367937,
"epoch": 0.4447351209941138,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9357602596282959,
"kl": 0.07438107710331679,
"learning_rate": 3.8897972531066054e-07,
"loss": 0.047136351466178894,
"num_tokens": 15272992.0,
"reward": -1.0043382800972722e-07,
"reward_std": 0.9973943531513214,
"rewards/rnd_reward/mean": -1.0043382658864175e-07,
"rewards/rnd_reward/std": 0.997394347190857,
"step": 680,
"step_time": 9.536431295378133
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.14375000223517417,
"completions/max_length": 384.0,
"completions/max_terminated_length": 352.2,
"completions/mean_length": 125.81125106811524,
"completions/mean_terminated_length": 82.36017532348633,
"completions/min_length": 2.3,
"completions/min_terminated_length": 2.3,
"entropy": 2.2503296345472337,
"epoch": 0.4512753433616743,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.7484872937202454,
"kl": 0.0611060356721282,
"learning_rate": 3.873446697187704e-07,
"loss": 0.09799213409423828,
"num_tokens": 15463906.0,
"reward": -6.347894991876046e-08,
"reward_std": 0.9972564399242401,
"rewards/rnd_reward/mean": -6.34789495634891e-08,
"rewards/rnd_reward/std": 0.9972564518451691,
"step": 690,
"step_time": 9.653547045495362
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15500000342726708,
"completions/max_length": 384.0,
"completions/max_terminated_length": 360.6,
"completions/mean_length": 133.82500076293945,
"completions/mean_terminated_length": 87.98081855773925,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.5280281245708465,
"epoch": 0.4578155657292348,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9658915996551514,
"kl": 0.059800021536648276,
"learning_rate": 3.857096141268803e-07,
"loss": 0.12680201530456542,
"num_tokens": 15667301.0,
"reward": -1.5869737026719122e-07,
"reward_std": 0.9973293662071228,
"rewards/rnd_reward/mean": -1.5869737026719122e-07,
"rewards/rnd_reward/std": 0.9973293721675873,
"step": 700,
"step_time": 9.966963570658118
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.10875000208616256,
"completions/max_length": 384.0,
"completions/max_terminated_length": 344.1,
"completions/mean_length": 107.25000228881837,
"completions/mean_terminated_length": 73.4811897277832,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.4395824640989305,
"epoch": 0.4643557880967953,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.049780011177063,
"kl": 0.08286946956068278,
"learning_rate": 3.840745585349902e-07,
"loss": 0.15116369724273682,
"num_tokens": 15847671.0,
"reward": -1.005828406341891e-07,
"reward_std": 0.9973811268806457,
"rewards/rnd_reward/mean": -1.0058283921310362e-07,
"rewards/rnd_reward/std": 0.9973811089992524,
"step": 710,
"step_time": 9.890902339830063
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09000000115484,
"completions/max_length": 384.0,
"completions/max_terminated_length": 345.2,
"completions/mean_length": 93.01625213623046,
"completions/mean_terminated_length": 64.09797096252441,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.8066621690988542,
"epoch": 0.4708960104643558,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.34576416015625,
"kl": 0.10443261573091149,
"learning_rate": 3.8243950294310006e-07,
"loss": 0.10605430603027344,
"num_tokens": 16021059.0,
"reward": -1.0505319210096787e-07,
"reward_std": 0.9972734034061432,
"rewards/rnd_reward/mean": -1.0505318357445503e-07,
"rewards/rnd_reward/std": 0.9972733795642853,
"step": 720,
"step_time": 10.098007511859759
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.08750000074505807,
"completions/max_length": 384.0,
"completions/max_terminated_length": 345.0,
"completions/mean_length": 94.07875137329101,
"completions/mean_terminated_length": 66.15557479858398,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.6466055154800414,
"epoch": 0.4774362328319163,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.0772963762283325,
"kl": 0.10096871191635728,
"learning_rate": 3.8080444735120994e-07,
"loss": 0.09338142871856689,
"num_tokens": 16191222.0,
"reward": -1.777708558137192e-07,
"reward_std": 0.9971938729286194,
"rewards/rnd_reward/mean": -1.7777085012937732e-07,
"rewards/rnd_reward/std": 0.9971938788890838,
"step": 730,
"step_time": 10.299412568542175
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.13375000208616256,
"completions/max_length": 384.0,
"completions/max_terminated_length": 365.1,
"completions/mean_length": 126.26250228881835,
"completions/mean_terminated_length": 86.43918380737304,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.173070195317268,
"epoch": 0.4839764551994768,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.7416156530380249,
"kl": 0.053471417538821696,
"learning_rate": 3.7916939175931976e-07,
"loss": 0.07696372270584106,
"num_tokens": 16393972.0,
"reward": -1.7151236602064303e-07,
"reward_std": 0.9972269594669342,
"rewards/rnd_reward/mean": -1.715123674417285e-07,
"rewards/rnd_reward/std": 0.9972269415855408,
"step": 740,
"step_time": 10.502406498114578
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.148750002682209,
"completions/max_length": 384.0,
"completions/max_terminated_length": 365.0,
"completions/mean_length": 139.12875213623047,
"completions/mean_terminated_length": 96.77134628295899,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.325762075185776,
"epoch": 0.49051667756703726,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.7352995872497559,
"kl": 0.05052672061137855,
"learning_rate": 3.775343361674297e-07,
"loss": 0.09521990418434143,
"num_tokens": 16596670.0,
"reward": -2.0787120362797397e-07,
"reward_std": 0.9972488164901734,
"rewards/rnd_reward/mean": -2.0787119723308933e-07,
"rewards/rnd_reward/std": 0.9972488105297088,
"step": 750,
"step_time": 9.698242208594456
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.14750000238418579,
"completions/max_length": 384.0,
"completions/max_terminated_length": 371.2,
"completions/mean_length": 127.9612533569336,
"completions/mean_terminated_length": 83.97822761535645,
"completions/min_length": 2.2,
"completions/min_terminated_length": 2.2,
"entropy": 2.2959645807743074,
"epoch": 0.4970568999345978,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8263393044471741,
"kl": 0.05697637004777789,
"learning_rate": 3.7589928057553957e-07,
"loss": 0.10516003370285035,
"num_tokens": 16792629.0,
"reward": -1.6942620391091622e-07,
"reward_std": 0.9972022175788879,
"rewards/rnd_reward/mean": -1.6942620106874529e-07,
"rewards/rnd_reward/std": 0.9972022175788879,
"step": 760,
"step_time": 10.044187037739903
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.10875000208616256,
"completions/max_length": 384.0,
"completions/max_terminated_length": 339.2,
"completions/mean_length": 112.1150016784668,
"completions/mean_terminated_length": 79.13344192504883,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.7552184253931045,
"epoch": 0.5035971223021583,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.029379963874817,
"kl": 0.07228619959205389,
"learning_rate": 3.742642249836494e-07,
"loss": 0.03081640601158142,
"num_tokens": 16971246.0,
"reward": -8.299946501200139e-08,
"reward_std": 0.9970763862133026,
"rewards/rnd_reward/mean": -8.299946785417234e-08,
"rewards/rnd_reward/std": 0.9970763862133026,
"step": 770,
"step_time": 9.667120404168964
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.13125000298023223,
"completions/max_length": 384.0,
"completions/max_terminated_length": 361.9,
"completions/mean_length": 124.3275032043457,
"completions/mean_terminated_length": 85.28903274536133,
"completions/min_length": 2.5,
"completions/min_terminated_length": 2.5,
"entropy": 2.8021608620882033,
"epoch": 0.5101373446697187,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.0165283679962158,
"kl": 0.07025760738179088,
"learning_rate": 3.7262916939175933e-07,
"loss": 0.03996403813362122,
"num_tokens": 17162308.0,
"reward": -5.170702825907369e-08,
"reward_std": 0.9971189498901367,
"rewards/rnd_reward/mean": -5.170703110124464e-08,
"rewards/rnd_reward/std": 0.9971189439296723,
"step": 780,
"step_time": 9.969513631938025
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.13375000283122063,
"completions/max_length": 384.0,
"completions/max_terminated_length": 348.5,
"completions/mean_length": 127.03500213623047,
"completions/mean_terminated_length": 87.38848876953125,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.637522208690643,
"epoch": 0.5166775670372793,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9886835813522339,
"kl": 0.06816753121092915,
"learning_rate": 3.7099411379986915e-07,
"loss": 0.07478119730949402,
"num_tokens": 17356536.0,
"reward": -1.1831521575800253e-07,
"reward_std": 0.9971611678600312,
"rewards/rnd_reward/mean": -1.1831521575800253e-07,
"rewards/rnd_reward/std": 0.997161191701889,
"step": 790,
"step_time": 9.748413591040299
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.12625000327825547,
"completions/max_length": 384.0,
"completions/max_terminated_length": 356.3,
"completions/mean_length": 120.86375045776367,
"completions/mean_terminated_length": 83.05458145141601,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.6706619918346406,
"epoch": 0.5232177894048398,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8250897526741028,
"kl": 0.07074365848675371,
"learning_rate": 3.693590582079791e-07,
"loss": 0.05061581134796143,
"num_tokens": 17545417.0,
"reward": -6.467104096685716e-08,
"reward_std": 0.9972768127918243,
"rewards/rnd_reward/mean": -6.467104096685716e-08,
"rewards/rnd_reward/std": 0.9972768068313599,
"step": 800,
"step_time": 9.678878564713523
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.12500000298023223,
"completions/max_length": 384.0,
"completions/max_terminated_length": 363.5,
"completions/mean_length": 121.30500183105468,
"completions/mean_terminated_length": 83.6633544921875,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.795678195357323,
"epoch": 0.5297580117724002,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.0277215242385864,
"kl": 0.06990557787939906,
"learning_rate": 3.677240026160889e-07,
"loss": 0.059202158451080324,
"num_tokens": 17737161.0,
"reward": -1.2487173535191686e-07,
"reward_std": 0.9970629572868347,
"rewards/rnd_reward/mean": -1.2487172682540403e-07,
"rewards/rnd_reward/std": 0.9970629572868347,
"step": 810,
"step_time": 10.510143173905089
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.13000000268220901,
"completions/max_length": 384.0,
"completions/max_terminated_length": 352.4,
"completions/mean_length": 119.66500091552734,
"completions/mean_terminated_length": 80.23575859069824,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.8059179961681364,
"epoch": 0.5362982341399608,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8309988379478455,
"kl": 0.08165676584467292,
"learning_rate": 3.660889470241988e-07,
"loss": 0.12505731582641602,
"num_tokens": 17927903.0,
"reward": -2.3901462826714236e-07,
"reward_std": 0.997013247013092,
"rewards/rnd_reward/mean": -2.3901462400388593e-07,
"rewards/rnd_reward/std": 0.9970132112503052,
"step": 820,
"step_time": 10.190648046904244
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.14875000417232515,
"completions/max_length": 384.0,
"completions/max_terminated_length": 360.9,
"completions/mean_length": 126.17125396728515,
"completions/mean_terminated_length": 80.60145225524903,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.938824012875557,
"epoch": 0.5428384565075213,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.1022851467132568,
"kl": 0.07984373616054655,
"learning_rate": 3.644538914323087e-07,
"loss": 0.023773609101772307,
"num_tokens": 18123110.0,
"reward": -2.2649766062698973e-08,
"reward_std": 0.9969716727733612,
"rewards/rnd_reward/mean": -2.2649762509985294e-08,
"rewards/rnd_reward/std": 0.9969716787338256,
"step": 830,
"step_time": 9.89921972535085
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.14375000447034836,
"completions/max_length": 384.0,
"completions/max_terminated_length": 356.7,
"completions/mean_length": 127.39125366210938,
"completions/mean_terminated_length": 84.2758415222168,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.8030223518610002,
"epoch": 0.5493786788750817,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9427273273468018,
"kl": 0.0753337662667036,
"learning_rate": 3.6281883584041854e-07,
"loss": 0.05915579795837402,
"num_tokens": 18319518.0,
"reward": -1.1295080284412507e-07,
"reward_std": 0.9969306707382202,
"rewards/rnd_reward/mean": -1.1295080142303959e-07,
"rewards/rnd_reward/std": 0.9969306766986847,
"step": 840,
"step_time": 9.87428822228685
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15750000178813933,
"completions/max_length": 384.0,
"completions/max_terminated_length": 368.2,
"completions/mean_length": 137.7587516784668,
"completions/mean_terminated_length": 91.88029098510742,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.810399681329727,
"epoch": 0.5559189012426422,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8602397441864014,
"kl": 0.0653490168042481,
"learning_rate": 3.611837802485284e-07,
"loss": 0.03035195469856262,
"num_tokens": 18526310.0,
"reward": -1.1295080284412507e-07,
"reward_std": 0.9969462454319,
"rewards/rnd_reward/mean": -1.1295080426521054e-07,
"rewards/rnd_reward/std": 0.9969462633132935,
"step": 850,
"step_time": 10.37288966330234
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15250000283122062,
"completions/max_length": 384.0,
"completions/max_terminated_length": 343.9,
"completions/mean_length": 127.73250350952148,
"completions/mean_terminated_length": 81.79599456787109,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.83120773434639,
"epoch": 0.5624591236102028,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.0732090473175049,
"kl": 0.08459939705207944,
"learning_rate": 3.595487246566383e-07,
"loss": 0.03491010665893555,
"num_tokens": 18724316.0,
"reward": -1.4752149617436318e-07,
"reward_std": 0.9969268858432769,
"rewards/rnd_reward/mean": -1.4752149759544864e-07,
"rewards/rnd_reward/std": 0.9969269037246704,
"step": 860,
"step_time": 10.52987401576247
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.14000000208616256,
"completions/max_length": 384.0,
"completions/max_terminated_length": 353.0,
"completions/mean_length": 131.95375213623046,
"completions/mean_terminated_length": 90.77078781127929,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.9562632203102113,
"epoch": 0.5689993459777632,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.969403862953186,
"kl": 0.08435474634170533,
"learning_rate": 3.579136690647482e-07,
"loss": 0.04807090759277344,
"num_tokens": 18926489.0,
"reward": -1.3545155894689742e-07,
"reward_std": 0.9968080937862396,
"rewards/rnd_reward/mean": -1.3545156178906837e-07,
"rewards/rnd_reward/std": 0.9968081057071686,
"step": 870,
"step_time": 9.85819350676611
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.16750000193715095,
"completions/max_length": 384.0,
"completions/max_terminated_length": 367.2,
"completions/mean_length": 137.6675033569336,
"completions/mean_terminated_length": 88.34969863891601,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.7675389796495438,
"epoch": 0.5755395683453237,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.828363835811615,
"kl": 0.07512732800096274,
"learning_rate": 3.5627861347285806e-07,
"loss": 0.010637739300727844,
"num_tokens": 19126833.0,
"reward": -7.003545690054125e-08,
"reward_std": 0.9967847049236298,
"rewards/rnd_reward/mean": -7.003545903216945e-08,
"rewards/rnd_reward/std": 0.9967847049236298,
"step": 880,
"step_time": 9.748134864587337
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.19125000312924384,
"completions/max_length": 384.0,
"completions/max_terminated_length": 361.5,
"completions/mean_length": 148.54750366210936,
"completions/mean_terminated_length": 93.13098068237305,
"completions/min_length": 2.5,
"completions/min_terminated_length": 2.5,
"entropy": 2.6605105638504027,
"epoch": 0.5820797907128843,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8364513516426086,
"kl": 0.06730008013546467,
"learning_rate": 3.5464355788096794e-07,
"loss": 0.0366318553686142,
"num_tokens": 19338336.0,
"reward": -1.0654330893089536e-07,
"reward_std": 0.9968020260334015,
"rewards/rnd_reward/mean": -1.0654329685166886e-07,
"rewards/rnd_reward/std": 0.9968020260334015,
"step": 890,
"step_time": 9.872815124248154
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.16625000312924385,
"completions/max_length": 384.0,
"completions/max_terminated_length": 347.5,
"completions/mean_length": 149.7400032043457,
"completions/mean_terminated_length": 103.11838226318359,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.535254544019699,
"epoch": 0.5886200130804448,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.7181042432785034,
"kl": 0.05683262627571821,
"learning_rate": 3.530085022890778e-07,
"loss": 0.08454052209854127,
"num_tokens": 19557363.0,
"reward": -1.737475452401327e-07,
"reward_std": 0.9967290103435517,
"rewards/rnd_reward/mean": -1.737475338714489e-07,
"rewards/rnd_reward/std": 0.9967289924621582,
"step": 900,
"step_time": 10.189786232402549
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1912500038743019,
"completions/max_length": 384.0,
"completions/max_terminated_length": 365.0,
"completions/mean_length": 157.67875213623046,
"completions/mean_terminated_length": 104.52761840820312,
"completions/min_length": 2.2,
"completions/min_terminated_length": 2.2,
"entropy": 2.5941266864538193,
"epoch": 0.5951602354480052,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9426244497299194,
"kl": 0.050556823750957845,
"learning_rate": 3.513734466971877e-07,
"loss": 0.07448753714561462,
"num_tokens": 19775536.0,
"reward": -2.162158537544201e-07,
"reward_std": 0.9968621492385864,
"rewards/rnd_reward/mean": -2.1621584735953548e-07,
"rewards/rnd_reward/std": 0.9968621671199799,
"step": 910,
"step_time": 9.74868165422231
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1662500023841858,
"completions/max_length": 384.0,
"completions/max_terminated_length": 363.0,
"completions/mean_length": 151.89000396728517,
"completions/mean_terminated_length": 105.61829071044922,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.508600893616676,
"epoch": 0.6017004578155657,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.7371200323104858,
"kl": 0.04880232191644609,
"learning_rate": 3.4973839110529757e-07,
"loss": 0.12514564990997315,
"num_tokens": 19990298.0,
"reward": -1.93715107066339e-08,
"reward_std": 0.9970090925693512,
"rewards/rnd_reward/mean": -1.937150893027706e-08,
"rewards/rnd_reward/std": 0.9970090866088868,
"step": 920,
"step_time": 9.937099263281562
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1575000025331974,
"completions/max_length": 384.0,
"completions/max_terminated_length": 355.5,
"completions/mean_length": 138.63375244140624,
"completions/mean_terminated_length": 92.52680130004883,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.4236575067043304,
"epoch": 0.6082406801831263,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9167628884315491,
"kl": 0.05787700763903558,
"learning_rate": 3.4810333551340745e-07,
"loss": 0.10835990905761719,
"num_tokens": 20187745.0,
"reward": -7.212161943215278e-08,
"reward_std": 0.996803218126297,
"rewards/rnd_reward/mean": -7.212161801106731e-08,
"rewards/rnd_reward/std": 0.9968032300472259,
"step": 930,
"step_time": 9.689079960295931
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1337500013411045,
"completions/max_length": 384.0,
"completions/max_terminated_length": 354.4,
"completions/mean_length": 129.22500305175782,
"completions/mean_terminated_length": 89.93161087036133,
"completions/min_length": 2.2,
"completions/min_terminated_length": 2.2,
"entropy": 2.6605860769748686,
"epoch": 0.6147809025506867,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.0838011503219604,
"kl": 0.06689512394368649,
"learning_rate": 3.4646827992151733e-07,
"loss": 0.07303244471549988,
"num_tokens": 20383990.0,
"reward": -1.1324883786301143e-08,
"reward_std": 0.9967760264873504,
"rewards/rnd_reward/mean": -1.1324883786301143e-08,
"rewards/rnd_reward/std": 0.9967760384082794,
"step": 940,
"step_time": 9.85847494918853
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.16125000268220901,
"completions/max_length": 384.0,
"completions/max_terminated_length": 364.1,
"completions/mean_length": 134.65625534057617,
"completions/mean_terminated_length": 86.58066329956054,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.7146200090646744,
"epoch": 0.6213211249182472,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9745354652404785,
"kl": 0.06564742000773549,
"learning_rate": 3.448332243296272e-07,
"loss": 0.10424284934997559,
"num_tokens": 20593190.0,
"reward": -1.0967255157368072e-07,
"reward_std": 0.9968843460083008,
"rewards/rnd_reward/mean": -1.096725451787961e-07,
"rewards/rnd_reward/std": 0.9968843400478363,
"step": 950,
"step_time": 10.862376277497969
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.10875000022351741,
"completions/max_length": 384.0,
"completions/max_terminated_length": 347.1,
"completions/mean_length": 113.23000411987304,
"completions/mean_terminated_length": 80.14122467041015,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.8509408473968505,
"epoch": 0.6278613472858077,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9323987364768982,
"kl": 0.08510139184072614,
"learning_rate": 3.4319816873773703e-07,
"loss": 0.0554282009601593,
"num_tokens": 20776164.0,
"reward": -7.957220322651893e-08,
"reward_std": 0.9968249201774597,
"rewards/rnd_reward/mean": -7.957220180543345e-08,
"rewards/rnd_reward/std": 0.9968249082565308,
"step": 960,
"step_time": 9.989999298867769
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.10750000327825546,
"completions/max_length": 384.0,
"completions/max_terminated_length": 355.2,
"completions/mean_length": 112.88375091552734,
"completions/mean_terminated_length": 80.20179443359375,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.88864361345768,
"epoch": 0.6344015696533682,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8851020932197571,
"kl": 0.0875691649504006,
"learning_rate": 3.4156311314584696e-07,
"loss": 0.10126043558120727,
"num_tokens": 20958826.0,
"reward": -1.6622245713193706e-07,
"reward_std": 0.9968612968921662,
"rewards/rnd_reward/mean": -1.6622244931596696e-07,
"rewards/rnd_reward/std": 0.9968612730503082,
"step": 970,
"step_time": 10.018015099223703
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.11125000156462192,
"completions/max_length": 384.0,
"completions/max_terminated_length": 353.6,
"completions/mean_length": 106.1625,
"completions/mean_terminated_length": 71.38758811950683,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.9656033277511598,
"epoch": 0.6409417920209287,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9377958178520203,
"kl": 0.09664760064333677,
"learning_rate": 3.3992805755395684e-07,
"loss": 0.046259531378746034,
"num_tokens": 21147866.0,
"reward": -2.499669804478799e-07,
"reward_std": 0.9968202471733093,
"rewards/rnd_reward/mean": -2.499669751188094e-07,
"rewards/rnd_reward/std": 0.9968202352523804,
"step": 980,
"step_time": 11.508573367865756
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1262500025331974,
"completions/max_length": 384.0,
"completions/max_terminated_length": 358.6,
"completions/mean_length": 122.56625289916992,
"completions/mean_terminated_length": 84.71166687011718,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.80318078994751,
"epoch": 0.6474820143884892,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9525420069694519,
"kl": 0.07310699988156558,
"learning_rate": 3.3829300196206667e-07,
"loss": 0.02322930097579956,
"num_tokens": 21338124.0,
"reward": -9.76771161731449e-08,
"reward_std": 0.9966861188411713,
"rewards/rnd_reward/mean": -9.7677110488803e-08,
"rewards/rnd_reward/std": 0.9966861248016358,
"step": 990,
"step_time": 9.876657200045884
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15375000201165676,
"completions/max_length": 384.0,
"completions/max_terminated_length": 359.8,
"completions/mean_length": 137.66875076293945,
"completions/mean_terminated_length": 92.6232063293457,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.9049280107021334,
"epoch": 0.6540222367560498,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8244759440422058,
"kl": 0.061428536847233775,
"learning_rate": 3.366579463701766e-07,
"loss": 0.05219428539276123,
"num_tokens": 21539579.0,
"reward": -1.6495585271059098e-07,
"reward_std": 0.9966946303844452,
"rewards/rnd_reward/mean": -1.6495585271059098e-07,
"rewards/rnd_reward/std": 0.9966946005821228,
"step": 1000,
"step_time": 9.666829459602013
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.18250000178813935,
"completions/max_length": 384.0,
"completions/max_terminated_length": 358.9,
"completions/mean_length": 146.93875427246093,
"completions/mean_terminated_length": 93.79228134155274,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.8016369342803955,
"epoch": 0.6605624591236102,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.766015350818634,
"kl": 0.05675441958010197,
"learning_rate": 3.350228907782864e-07,
"loss": 0.08666471242904664,
"num_tokens": 21754125.0,
"reward": -2.2858381782953073e-07,
"reward_std": 0.996807062625885,
"rewards/rnd_reward/mean": -2.2858381640844527e-07,
"rewards/rnd_reward/std": 0.9968070745468139,
"step": 1010,
"step_time": 10.005978721287102
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.13875000253319741,
"completions/max_length": 384.0,
"completions/max_terminated_length": 357.3,
"completions/mean_length": 136.35500259399413,
"completions/mean_terminated_length": 96.46944885253906,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.8755707025527952,
"epoch": 0.6671026814911707,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.0061216354370117,
"kl": 0.05596274421550333,
"learning_rate": 3.3338783518639635e-07,
"loss": 0.09954084157943725,
"num_tokens": 21957634.0,
"reward": -1.9520521661320344e-07,
"reward_std": 0.9966324806213379,
"rewards/rnd_reward/mean": -1.9520520950777608e-07,
"rewards/rnd_reward/std": 0.9966325044631958,
"step": 1020,
"step_time": 10.040163300000131
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.15875000432133674,
"completions/max_length": 384.0,
"completions/max_terminated_length": 358.5,
"completions/mean_length": 136.7437530517578,
"completions/mean_terminated_length": 90.20843048095703,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.7692076802253722,
"epoch": 0.6736429038587312,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8324238061904907,
"kl": 0.057948453910648824,
"learning_rate": 3.317527795945062e-07,
"loss": 0.07627512216567993,
"num_tokens": 22158654.0,
"reward": -3.7699937749380295e-08,
"reward_std": 0.9966008186340332,
"rewards/rnd_reward/mean": -3.769993632829482e-08,
"rewards/rnd_reward/std": 0.9966008424758911,
"step": 1030,
"step_time": 10.1797529961681
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.16250000298023223,
"completions/max_length": 384.0,
"completions/max_terminated_length": 362.7,
"completions/mean_length": 139.57500534057618,
"completions/mean_terminated_length": 92.21996688842773,
"completions/min_length": 2.2,
"completions/min_terminated_length": 2.2,
"entropy": 2.8831230252981186,
"epoch": 0.6801831262262917,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8833721876144409,
"kl": 0.060850747767835854,
"learning_rate": 3.3011772400261606e-07,
"loss": 0.10484603643417359,
"num_tokens": 22366559.0,
"reward": -1.0505318579490108e-07,
"reward_std": 0.996761429309845,
"rewards/rnd_reward/mean": -1.0505318588371893e-07,
"rewards/rnd_reward/std": 0.9967614591121674,
"step": 1040,
"step_time": 10.081386449630372
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1625000037252903,
"completions/max_length": 384.0,
"completions/max_terminated_length": 362.9,
"completions/mean_length": 137.33875198364257,
"completions/mean_terminated_length": 89.72673568725585,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.885282647609711,
"epoch": 0.6867233485938522,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9477590918540955,
"kl": 0.05932593122124672,
"learning_rate": 3.28482668410726e-07,
"loss": 0.024539683759212495,
"num_tokens": 22571575.0,
"reward": 4.9620866704458424e-08,
"reward_std": 0.996524053812027,
"rewards/rnd_reward/mean": 4.9620866704458424e-08,
"rewards/rnd_reward/std": 0.9965240597724915,
"step": 1050,
"step_time": 10.535549363144673
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.16250000447034835,
"completions/max_length": 384.0,
"completions/max_terminated_length": 357.5,
"completions/mean_length": 138.3237533569336,
"completions/mean_terminated_length": 90.70624694824218,
"completions/min_length": 2.2,
"completions/min_terminated_length": 2.2,
"entropy": 2.875243663787842,
"epoch": 0.6932635709614127,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9407637119293213,
"kl": 0.06043377108871937,
"learning_rate": 3.268476128188358e-07,
"loss": 0.06624546647071838,
"num_tokens": 22771314.0,
"reward": -5.781650713743147e-08,
"reward_std": 0.9966591358184814,
"rewards/rnd_reward/mean": -5.7816505716346e-08,
"rewards/rnd_reward/std": 0.9966591358184814,
"step": 1060,
"step_time": 10.021701269014738
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1400000039488077,
"completions/max_length": 384.0,
"completions/max_terminated_length": 342.1,
"completions/mean_length": 127.3300048828125,
"completions/mean_terminated_length": 85.42830429077148,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.927751249074936,
"epoch": 0.6998037933289731,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8508753180503845,
"kl": 0.0620661674067378,
"learning_rate": 3.252125572269457e-07,
"loss": 0.12821773290634156,
"num_tokens": 22965548.0,
"reward": -1.0244548391824537e-07,
"reward_std": 0.9966984510421752,
"rewards/rnd_reward/mean": -1.0244548391824537e-07,
"rewards/rnd_reward/std": 0.9966984808444976,
"step": 1070,
"step_time": 9.767576195369475
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.14500000327825546,
"completions/max_length": 384.0,
"completions/max_terminated_length": 363.6,
"completions/mean_length": 136.3862518310547,
"completions/mean_terminated_length": 94.67120056152343,
"completions/min_length": 2.5,
"completions/min_terminated_length": 2.5,
"entropy": 2.6616650968790054,
"epoch": 0.7063440156965337,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.789601743221283,
"kl": 0.0545062692835927,
"learning_rate": 3.2357750163505557e-07,
"loss": 0.11497254371643066,
"num_tokens": 23170007.0,
"reward": 1.226365569095833e-07,
"reward_std": 0.9966248035430908,
"rewards/rnd_reward/mean": 1.2263655548849784e-07,
"rewards/rnd_reward/std": 0.9966248333454132,
"step": 1080,
"step_time": 9.882125865970739
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.14250000566244125,
"completions/max_length": 384.0,
"completions/max_terminated_length": 356.8,
"completions/mean_length": 130.01000289916993,
"completions/mean_terminated_length": 87.69976119995117,
"completions/min_length": 2.3,
"completions/min_terminated_length": 2.3,
"entropy": 2.890427288413048,
"epoch": 0.7128842380640942,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8021807670593262,
"kl": 0.05592884449288249,
"learning_rate": 3.2194244604316545e-07,
"loss": 0.09672708511352539,
"num_tokens": 23366425.0,
"reward": 6.273389110411643e-08,
"reward_std": 0.9965825915336609,
"rewards/rnd_reward/mean": 6.273389181465916e-08,
"rewards/rnd_reward/std": 0.9965825736522674,
"step": 1090,
"step_time": 9.854898347100242
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.14500000178813935,
"completions/max_length": 384.0,
"completions/max_terminated_length": 336.9,
"completions/mean_length": 131.98875274658204,
"completions/mean_terminated_length": 89.18458480834961,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.7609634190797805,
"epoch": 0.7194244604316546,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.0092402696609497,
"kl": 0.05501702083274722,
"learning_rate": 3.2030739045127533e-07,
"loss": 0.08695750832557678,
"num_tokens": 23567981.0,
"reward": -1.3828277736216422e-07,
"reward_std": 0.9966061592102051,
"rewards/rnd_reward/mean": -1.3828277451999326e-07,
"rewards/rnd_reward/std": 0.9966061592102051,
"step": 1100,
"step_time": 10.145819194451906
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1212500013411045,
"completions/max_length": 384.0,
"completions/max_terminated_length": 362.9,
"completions/mean_length": 121.10875091552734,
"completions/mean_terminated_length": 84.89219970703125,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.8415472239255903,
"epoch": 0.7259646827992152,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8949585556983948,
"kl": 0.06347150038927793,
"learning_rate": 3.186723348593852e-07,
"loss": 0.08753393888473511,
"num_tokens": 23758168.0,
"reward": -1.1354684570363816e-07,
"reward_std": 0.9964610040187836,
"rewards/rnd_reward/mean": -1.1354684286146721e-07,
"rewards/rnd_reward/std": 0.9964609861373901,
"step": 1110,
"step_time": 10.006089114071802
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.12125000208616257,
"completions/max_length": 384.0,
"completions/max_terminated_length": 369.7,
"completions/mean_length": 121.20000228881835,
"completions/mean_terminated_length": 84.96716842651367,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.9164877772331237,
"epoch": 0.7325049051667757,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.1227521896362305,
"kl": 0.06196983400732279,
"learning_rate": 3.170372792674951e-07,
"loss": 0.07237975597381592,
"num_tokens": 23955023.0,
"reward": -5.5432321488524396e-08,
"reward_std": 0.9964903295040131,
"rewards/rnd_reward/mean": -5.543231935689619e-08,
"rewards/rnd_reward/std": 0.9964903354644775,
"step": 1120,
"step_time": 10.375506636733189
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.11875000298023224,
"completions/max_length": 384.0,
"completions/max_terminated_length": 356.2,
"completions/mean_length": 117.09375,
"completions/mean_terminated_length": 81.17416534423828,
"completions/min_length": 2.2,
"completions/min_terminated_length": 2.2,
"entropy": 2.7361535727977753,
"epoch": 0.7390451275343362,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8529885411262512,
"kl": 0.06673006983473898,
"learning_rate": 3.1540222367560496e-07,
"loss": 0.09247955083847045,
"num_tokens": 24140203.0,
"reward": -2.6300548938706926e-08,
"reward_std": 0.9966336786746979,
"rewards/rnd_reward/mean": -2.6300551070335132e-08,
"rewards/rnd_reward/std": 0.9966336965560914,
"step": 1130,
"step_time": 10.012891574040987
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.11500000357627868,
"completions/max_length": 384.0,
"completions/max_terminated_length": 351.7,
"completions/mean_length": 119.6712516784668,
"completions/mean_terminated_length": 85.35453567504882,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.6402025282382966,
"epoch": 0.7455853499018966,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9655517935752869,
"kl": 0.06339178960770368,
"learning_rate": 3.1376716808371484e-07,
"loss": 0.012396740168333054,
"num_tokens": 24333505.0,
"reward": -1.3798475424486866e-07,
"reward_std": 0.9963988780975341,
"rewards/rnd_reward/mean": -1.379847538895973e-07,
"rewards/rnd_reward/std": 0.9963988780975341,
"step": 1140,
"step_time": 10.81275618087966
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.12750000208616258,
"completions/max_length": 384.0,
"completions/max_terminated_length": 358.4,
"completions/mean_length": 130.15375289916992,
"completions/mean_terminated_length": 93.11451568603516,
"completions/min_length": 2.6,
"completions/min_terminated_length": 2.6,
"entropy": 2.7988665878772734,
"epoch": 0.7521255722694572,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9594887495040894,
"kl": 0.057864805683493614,
"learning_rate": 3.121321124918247e-07,
"loss": 0.05015655755996704,
"num_tokens": 24526988.0,
"reward": -2.7045605088815705e-08,
"reward_std": 0.9962463498115539,
"rewards/rnd_reward/mean": -2.7045605088815705e-08,
"rewards/rnd_reward/std": 0.9962463617324829,
"step": 1150,
"step_time": 9.847500793170184
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.13000000305473805,
"completions/max_length": 384.0,
"completions/max_terminated_length": 340.1,
"completions/mean_length": 119.13750076293945,
"completions/mean_terminated_length": 79.57122192382812,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.799566090106964,
"epoch": 0.7586657946370177,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.3144031763076782,
"kl": 0.06464958414435387,
"learning_rate": 3.104970568999346e-07,
"loss": 0.08081794381141663,
"num_tokens": 24712093.0,
"reward": 2.2947787847726885e-08,
"reward_std": 0.9963953137397766,
"rewards/rnd_reward/mean": 2.2947789268812358e-08,
"rewards/rnd_reward/std": 0.9963953375816346,
"step": 1160,
"step_time": 9.831479465053416
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.13000000268220901,
"completions/max_length": 384.0,
"completions/max_terminated_length": 366.1,
"completions/mean_length": 128.67500228881835,
"completions/mean_terminated_length": 90.44079208374023,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.758173832297325,
"epoch": 0.7652060170045781,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8444108963012695,
"kl": 0.061153271794319154,
"learning_rate": 3.088620013080445e-07,
"loss": 0.06105284690856934,
"num_tokens": 24905938.0,
"reward": -9.223818580039733e-08,
"reward_std": 0.9962478995323181,
"rewards/rnd_reward/mean": -9.22381850898546e-08,
"rewards/rnd_reward/std": 0.9962479054927826,
"step": 1170,
"step_time": 9.715631593717262
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.13125000149011612,
"completions/max_length": 384.0,
"completions/max_terminated_length": 367.7,
"completions/mean_length": 132.65000381469727,
"completions/mean_terminated_length": 94.60303192138672,
"completions/min_length": 2.2,
"completions/min_terminated_length": 2.2,
"entropy": 2.652452102303505,
"epoch": 0.7717462393721386,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8246821761131287,
"kl": 0.05974919479340315,
"learning_rate": 3.072269457161543e-07,
"loss": 0.07230232954025269,
"num_tokens": 25111683.0,
"reward": -1.4752150079289095e-08,
"reward_std": 0.9964145243167877,
"rewards/rnd_reward/mean": -1.4752149368746359e-08,
"rewards/rnd_reward/std": 0.9964145421981812,
"step": 1180,
"step_time": 10.24143796269782
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1437500037252903,
"completions/max_length": 384.0,
"completions/max_terminated_length": 351.8,
"completions/mean_length": 133.8912498474121,
"completions/mean_terminated_length": 92.00172958374023,
"completions/min_length": 2.2,
"completions/min_terminated_length": 2.2,
"entropy": 2.534428471326828,
"epoch": 0.7782864617396992,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.7489418387413025,
"kl": 0.05822144625708461,
"learning_rate": 3.0559189012426423e-07,
"loss": 0.07474024295806884,
"num_tokens": 25319036.0,
"reward": -8.940696432091499e-08,
"reward_std": 0.9965973794460297,
"rewards/rnd_reward/mean": -8.940696574200047e-08,
"rewards/rnd_reward/std": 0.9965973973274231,
"step": 1190,
"step_time": 10.355807177536189
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1450000025331974,
"completions/max_length": 384.0,
"completions/max_terminated_length": 365.6,
"completions/mean_length": 138.9825012207031,
"completions/mean_terminated_length": 97.18752059936523,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.8433078229427338,
"epoch": 0.7848266841072596,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8156785368919373,
"kl": 0.06143994340673089,
"learning_rate": 3.039568345323741e-07,
"loss": 0.08815957307815551,
"num_tokens": 25526837.0,
"reward": -1.0192394315211572e-07,
"reward_std": 0.996451473236084,
"rewards/rnd_reward/mean": -1.0192393888885931e-07,
"rewards/rnd_reward/std": 0.996451485157013,
"step": 1200,
"step_time": 10.298059746343643
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.14000000283122063,
"completions/max_length": 384.0,
"completions/max_terminated_length": 355.0,
"completions/mean_length": 127.81375274658203,
"completions/mean_terminated_length": 85.73782577514649,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.7779837220907213,
"epoch": 0.7913669064748201,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.937714695930481,
"kl": 0.06527232667431235,
"learning_rate": 3.0232177894048394e-07,
"loss": 0.08186920881271362,
"num_tokens": 25721938.0,
"reward": -1.0982156268823929e-07,
"reward_std": 0.9962611317634582,
"rewards/rnd_reward/mean": -1.0982156055661107e-07,
"rewards/rnd_reward/std": 0.9962611377239228,
"step": 1210,
"step_time": 9.923247428331525
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.13625000193715095,
"completions/max_length": 384.0,
"completions/max_terminated_length": 361.9,
"completions/mean_length": 128.51625213623046,
"completions/mean_terminated_length": 88.076513671875,
"completions/min_length": 2.3,
"completions/min_terminated_length": 2.3,
"entropy": 2.661186295747757,
"epoch": 0.7979071288423807,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9877737164497375,
"kl": 0.06118576750159264,
"learning_rate": 3.0068672334859387e-07,
"loss": 0.0857498049736023,
"num_tokens": 25916611.0,
"reward": -9.454786980001017e-08,
"reward_std": 0.9962615430355072,
"rewards/rnd_reward/mean": -9.454786553675376e-08,
"rewards/rnd_reward/std": 0.9962615609169007,
"step": 1220,
"step_time": 10.026552183739842
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.13875000402331353,
"completions/max_length": 384.0,
"completions/max_terminated_length": 358.7,
"completions/mean_length": 127.41250381469726,
"completions/mean_terminated_length": 86.15239715576172,
"completions/min_length": 2.2,
"completions/min_terminated_length": 2.2,
"entropy": 2.597662454843521,
"epoch": 0.8044473512099412,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9214726090431213,
"kl": 0.06120588518679142,
"learning_rate": 2.990516677567037e-07,
"loss": 0.08460624814033509,
"num_tokens": 26120156.0,
"reward": -1.5325844344715732e-07,
"reward_std": 0.9961971819400788,
"rewards/rnd_reward/mean": -1.5325843989444364e-07,
"rewards/rnd_reward/std": 0.9961971819400788,
"step": 1230,
"step_time": 10.459681713976897
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.12875000089406968,
"completions/max_length": 384.0,
"completions/max_terminated_length": 347.9,
"completions/mean_length": 121.05875167846679,
"completions/mean_terminated_length": 82.47761840820313,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.899898463487625,
"epoch": 0.8109875735775016,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9018652439117432,
"kl": 0.06602321406826377,
"learning_rate": 2.9741661216481357e-07,
"loss": 0.05030661225318909,
"num_tokens": 26307033.0,
"reward": -1.496076620810527e-07,
"reward_std": 0.996212637424469,
"rewards/rnd_reward/mean": -1.4960765781779628e-07,
"rewards/rnd_reward/std": 0.9962126433849334,
"step": 1240,
"step_time": 9.930500056478195
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.13875000253319741,
"completions/max_length": 384.0,
"completions/max_terminated_length": 359.1,
"completions/mean_length": 125.87625274658203,
"completions/mean_terminated_length": 84.06948699951172,
"completions/min_length": 2.2,
"completions/min_terminated_length": 2.2,
"entropy": 2.7101810783147813,
"epoch": 0.8175277959450621,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9933876991271973,
"kl": 0.06315780784934759,
"learning_rate": 2.9578155657292345e-07,
"loss": 0.14283275604248047,
"num_tokens": 26506699.0,
"reward": -4.701316604904804e-08,
"reward_std": 0.996234393119812,
"rewards/rnd_reward/mean": -4.7013166404319404e-08,
"rewards/rnd_reward/std": 0.9962343990802764,
"step": 1250,
"step_time": 11.001938613550738
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.14875000342726707,
"completions/max_length": 384.0,
"completions/max_terminated_length": 365.5,
"completions/mean_length": 131.21625289916992,
"completions/mean_terminated_length": 87.06982650756837,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.695519116520882,
"epoch": 0.8240680183126227,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8424022793769836,
"kl": 0.06466643353924155,
"learning_rate": 2.9414650098103333e-07,
"loss": 0.08369569182395935,
"num_tokens": 26703352.0,
"reward": -2.6673080633088375e-08,
"reward_std": 0.9962631404399872,
"rewards/rnd_reward/mean": -2.6673077790917433e-08,
"rewards/rnd_reward/std": 0.9962631464004517,
"step": 1260,
"step_time": 9.866059662331827
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.16375000178813934,
"completions/max_length": 384.0,
"completions/max_terminated_length": 368.6,
"completions/mean_length": 132.26625137329103,
"completions/mean_terminated_length": 83.00667343139648,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.7214665055274962,
"epoch": 0.8306082406801831,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.061651349067688,
"kl": 0.060832435730844736,
"learning_rate": 2.9251144538914326e-07,
"loss": 0.011636599898338318,
"num_tokens": 26905195.0,
"reward": -1.28149986178272e-08,
"reward_std": 0.996177339553833,
"rewards/rnd_reward/mean": -1.2814998262555832e-08,
"rewards/rnd_reward/std": 0.9961773693561554,
"step": 1270,
"step_time": 10.226913845306262
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1462500013411045,
"completions/max_length": 384.0,
"completions/max_terminated_length": 345.4,
"completions/mean_length": 135.36000289916993,
"completions/mean_terminated_length": 92.69313278198243,
"completions/min_length": 2.2,
"completions/min_terminated_length": 2.2,
"entropy": 2.5201783299446108,
"epoch": 0.8371484630477436,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9056648015975952,
"kl": 0.05058280224911869,
"learning_rate": 2.908763897972531e-07,
"loss": 0.08019465208053589,
"num_tokens": 27106543.0,
"reward": -1.9341707613307335e-07,
"reward_std": 0.9962323784828186,
"rewards/rnd_reward/mean": -1.9341706973818874e-07,
"rewards/rnd_reward/std": 0.9962323844432831,
"step": 1280,
"step_time": 10.01526041161269
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1225000001490116,
"completions/max_length": 384.0,
"completions/max_terminated_length": 373.1,
"completions/mean_length": 130.23250198364258,
"completions/mean_terminated_length": 94.88236541748047,
"completions/min_length": 2.3,
"completions/min_terminated_length": 2.3,
"entropy": 2.623024606704712,
"epoch": 0.8436886854153042,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.783699095249176,
"kl": 0.05370175880379975,
"learning_rate": 2.8924133420536296e-07,
"loss": 0.11414051055908203,
"num_tokens": 27305554.0,
"reward": -9.253621175986382e-08,
"reward_std": 0.9961644470691681,
"rewards/rnd_reward/mean": -9.253620731897172e-08,
"rewards/rnd_reward/std": 0.9961644232273101,
"step": 1290,
"step_time": 10.230407970934175
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.14875000193715096,
"completions/max_length": 384.0,
"completions/max_terminated_length": 365.6,
"completions/mean_length": 136.5575035095215,
"completions/mean_terminated_length": 93.4459114074707,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.676475775241852,
"epoch": 0.8502289077828646,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.0168393850326538,
"kl": 0.052173000667244195,
"learning_rate": 2.8760627861347284e-07,
"loss": 0.049798119068145755,
"num_tokens": 27507145.0,
"reward": -1.576542894232702e-07,
"reward_std": 0.9962258338928223,
"rewards/rnd_reward/mean": -1.5765429013381292e-07,
"rewards/rnd_reward/std": 0.9962258338928223,
"step": 1300,
"step_time": 9.98130206335336
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.13500000163912773,
"completions/max_length": 384.0,
"completions/max_terminated_length": 354.1,
"completions/mean_length": 128.2087516784668,
"completions/mean_terminated_length": 88.18164749145508,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.640226128697395,
"epoch": 0.8567691301504251,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8835275769233704,
"kl": 0.05669898777268827,
"learning_rate": 2.859712230215827e-07,
"loss": 0.10237141847610473,
"num_tokens": 27707452.0,
"reward": -6.869435509315736e-08,
"reward_std": 0.9961890399456024,
"rewards/rnd_reward/mean": -6.869434940881546e-08,
"rewards/rnd_reward/std": 0.99618901014328,
"step": 1310,
"step_time": 10.37953436721582
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.13625000268220902,
"completions/max_length": 384.0,
"completions/max_terminated_length": 355.5,
"completions/mean_length": 128.80750274658203,
"completions/mean_terminated_length": 88.5141227722168,
"completions/min_length": 2.2,
"completions/min_terminated_length": 2.2,
"entropy": 2.4644729048013687,
"epoch": 0.8633093525179856,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8116011619567871,
"kl": 0.05524417916312814,
"learning_rate": 2.843361674296926e-07,
"loss": 0.1121416449546814,
"num_tokens": 27900098.0,
"reward": -5.230307502657183e-08,
"reward_std": 0.9961800217628479,
"rewards/rnd_reward/mean": -5.230307715820004e-08,
"rewards/rnd_reward/std": 0.9961800217628479,
"step": 1320,
"step_time": 9.958397940779104
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.11625000089406967,
"completions/max_length": 384.0,
"completions/max_terminated_length": 358.3,
"completions/mean_length": 122.20125198364258,
"completions/mean_terminated_length": 87.82691497802735,
"completions/min_length": 2.3,
"completions/min_terminated_length": 2.3,
"entropy": 2.619413223862648,
"epoch": 0.8698495748855462,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.1175379753112793,
"kl": 0.06109289322048426,
"learning_rate": 2.827011118378025e-07,
"loss": 0.05299858450889587,
"num_tokens": 28083799.0,
"reward": -9.030103811369373e-08,
"reward_std": 0.9959750473499298,
"rewards/rnd_reward/mean": -9.030103793605804e-08,
"rewards/rnd_reward/std": 0.9959750413894654,
"step": 1330,
"step_time": 9.82245781556703
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.11625000312924386,
"completions/max_length": 384.0,
"completions/max_terminated_length": 361.4,
"completions/mean_length": 122.17500152587891,
"completions/mean_terminated_length": 87.70087432861328,
"completions/min_length": 2.2,
"completions/min_terminated_length": 2.2,
"entropy": 2.900624880194664,
"epoch": 0.8763897972531066,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.942399263381958,
"kl": 0.0663422117009759,
"learning_rate": 2.8106605624591235e-07,
"loss": 0.07738288044929505,
"num_tokens": 28270204.0,
"reward": -1.4573336244438905e-07,
"reward_std": 0.9961058497428894,
"rewards/rnd_reward/mean": -1.4573335960221812e-07,
"rewards/rnd_reward/std": 0.9961058616638183,
"step": 1340,
"step_time": 9.953787335986272
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1312500022351742,
"completions/max_length": 384.0,
"completions/max_terminated_length": 370.4,
"completions/mean_length": 120.04500045776368,
"completions/mean_terminated_length": 80.18856277465821,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.87275829911232,
"epoch": 0.8829300196206671,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.0761396884918213,
"kl": 0.07089657634496689,
"learning_rate": 2.794310006540222e-07,
"loss": 0.13924288749694824,
"num_tokens": 28461485.0,
"reward": 2.965331162840812e-08,
"reward_std": 0.9960719764232635,
"rewards/rnd_reward/mean": 2.965331162840812e-08,
"rewards/rnd_reward/std": 0.996071970462799,
"step": 1350,
"step_time": 10.60484689238947
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.12875000163912773,
"completions/max_length": 384.0,
"completions/max_terminated_length": 340.8,
"completions/mean_length": 115.14375228881836,
"completions/mean_terminated_length": 75.8135986328125,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 3.0420612663030626,
"epoch": 0.8894702419882276,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.1144483089447021,
"kl": 0.07124942755326628,
"learning_rate": 2.777959450621321e-07,
"loss": 0.08705209493637085,
"num_tokens": 28652205.0,
"reward": -1.2695789215477759e-07,
"reward_std": 0.9961645841598511,
"rewards/rnd_reward/mean": -1.2695789215477759e-07,
"rewards/rnd_reward/std": 0.99616459608078,
"step": 1360,
"step_time": 10.374728590669111
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.12250000275671483,
"completions/max_length": 384.0,
"completions/max_terminated_length": 347.6,
"completions/mean_length": 116.32750015258789,
"completions/mean_terminated_length": 78.6882926940918,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.8743550151586534,
"epoch": 0.8960104643557881,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.038732886314392,
"kl": 0.07000278439372778,
"learning_rate": 2.76160889470242e-07,
"loss": 0.11522959470748902,
"num_tokens": 28836007.0,
"reward": -2.1174550361990897e-07,
"reward_std": 0.9960576713085174,
"rewards/rnd_reward/mean": -2.117455032646376e-07,
"rewards/rnd_reward/std": 0.9960576832294464,
"step": 1370,
"step_time": 10.044307312183083
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.10750000178813934,
"completions/max_length": 384.0,
"completions/max_terminated_length": 354.4,
"completions/mean_length": 112.57625274658203,
"completions/mean_terminated_length": 79.90474357604981,
"completions/min_length": 2.2,
"completions/min_terminated_length": 2.2,
"entropy": 2.814165809750557,
"epoch": 0.9025506867233486,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.853209912776947,
"kl": 0.07179404897615313,
"learning_rate": 2.7452583387835187e-07,
"loss": 0.11325899362564087,
"num_tokens": 29020798.0,
"reward": -4.172328260665381e-09,
"reward_std": 0.995877069234848,
"rewards/rnd_reward/mean": -4.172324707951702e-09,
"rewards/rnd_reward/std": 0.9958770573139191,
"step": 1380,
"step_time": 10.452752491505816
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09500000216066837,
"completions/max_length": 384.0,
"completions/max_terminated_length": 359.1,
"completions/mean_length": 97.97000122070312,
"completions/mean_terminated_length": 68.24239921569824,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.9521768242120743,
"epoch": 0.9090909090909091,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.2329504489898682,
"kl": 0.09182704174891114,
"learning_rate": 2.7289077828646175e-07,
"loss": 0.09825693368911743,
"num_tokens": 29195334.0,
"reward": 9.715557460765468e-08,
"reward_std": 0.9959101438522339,
"rewards/rnd_reward/mean": 9.715556679168457e-08,
"rewards/rnd_reward/std": 0.9959101438522339,
"step": 1390,
"step_time": 10.322968539246357
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.10125000234693289,
"completions/max_length": 384.0,
"completions/max_terminated_length": 353.9,
"completions/mean_length": 103.2087516784668,
"completions/mean_terminated_length": 71.93345756530762,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 3.0182015657424928,
"epoch": 0.9156311314584696,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.4850752353668213,
"kl": 0.09229131015017629,
"learning_rate": 2.7125572269457157e-07,
"loss": 0.07991544604301452,
"num_tokens": 29371486.0,
"reward": -1.5191734150654668e-07,
"reward_std": 0.995727401971817,
"rewards/rnd_reward/mean": -1.5191734008546122e-07,
"rewards/rnd_reward/std": 0.9957274079322815,
"step": 1400,
"step_time": 10.564200303750113
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.11875000409781933,
"completions/max_length": 384.0,
"completions/max_terminated_length": 352.6,
"completions/mean_length": 109.10750198364258,
"completions/mean_terminated_length": 72.09171180725097,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.9613120049238204,
"epoch": 0.9221713538260301,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9964513778686523,
"kl": 0.0881183447316289,
"learning_rate": 2.696206671026815e-07,
"loss": 0.05779829025268555,
"num_tokens": 29547567.0,
"reward": -2.881884620276054e-07,
"reward_std": 0.9955705881118775,
"rewards/rnd_reward/mean": -2.881884591854345e-07,
"rewards/rnd_reward/std": 0.9955705881118775,
"step": 1410,
"step_time": 10.214018813963047
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.08625000156462193,
"completions/max_length": 384.0,
"completions/max_terminated_length": 348.2,
"completions/mean_length": 102.86875228881836,
"completions/mean_terminated_length": 76.34417495727538,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.703101697564125,
"epoch": 0.9287115761935906,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8821516633033752,
"kl": 0.08714565662667155,
"learning_rate": 2.679856115107914e-07,
"loss": 0.017349694669246674,
"num_tokens": 29720227.0,
"reward": -8.821487531562867e-08,
"reward_std": 0.9958377420902252,
"rewards/rnd_reward/mean": -8.821487389454319e-08,
"rewards/rnd_reward/std": 0.9958377361297608,
"step": 1420,
"step_time": 10.117785238358191
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.13000000044703483,
"completions/max_length": 384.0,
"completions/max_terminated_length": 347.4,
"completions/mean_length": 115.6462516784668,
"completions/mean_terminated_length": 75.65214309692382,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.66856093108654,
"epoch": 0.935251798561151,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8306900858879089,
"kl": 0.07128150118514895,
"learning_rate": 2.663505559189012e-07,
"loss": 0.07572194337844848,
"num_tokens": 29907299.0,
"reward": -1.2844801631217705e-07,
"reward_std": 0.995860880613327,
"rewards/rnd_reward/mean": -1.2844801062783516e-07,
"rewards/rnd_reward/std": 0.9958608686923981,
"step": 1430,
"step_time": 10.222553351335227
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.12500000223517418,
"completions/max_length": 384.0,
"completions/max_terminated_length": 346.7,
"completions/mean_length": 121.33125076293945,
"completions/mean_terminated_length": 83.87310562133788,
"completions/min_length": 2.3,
"completions/min_terminated_length": 2.3,
"entropy": 2.7864604860544206,
"epoch": 0.9417920209287116,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.97917640209198,
"kl": 0.06842925064265729,
"learning_rate": 2.6471550032701114e-07,
"loss": 0.11840918064117431,
"num_tokens": 30100959.0,
"reward": -1.5377998749954713e-07,
"reward_std": 0.9958663046360016,
"rewards/rnd_reward/mean": -1.5377998110466252e-07,
"rewards/rnd_reward/std": 0.9958662986755371,
"step": 1440,
"step_time": 10.766110814665444
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.14125000238418578,
"completions/max_length": 384.0,
"completions/max_terminated_length": 355.3,
"completions/mean_length": 124.87875289916992,
"completions/mean_terminated_length": 82.31183624267578,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.699194425344467,
"epoch": 0.9483322432962721,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8881458640098572,
"kl": 0.06711579142138362,
"learning_rate": 2.6308044473512096e-07,
"loss": 0.10254174470901489,
"num_tokens": 30297562.0,
"reward": -6.161630330403512e-08,
"reward_std": 0.9957123577594758,
"rewards/rnd_reward/mean": -6.16162979749646e-08,
"rewards/rnd_reward/std": 0.9957123517990112,
"step": 1450,
"step_time": 10.593307360098697
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1200000025331974,
"completions/max_length": 384.0,
"completions/max_terminated_length": 355.1,
"completions/mean_length": 116.06000213623047,
"completions/mean_terminated_length": 79.77308044433593,
"completions/min_length": 2.2,
"completions/min_terminated_length": 2.2,
"entropy": 2.995149564743042,
"epoch": 0.9548724656638325,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.0435981750488281,
"kl": 0.07414040835574269,
"learning_rate": 2.6144538914323084e-07,
"loss": 0.08301687836647034,
"num_tokens": 30483565.0,
"reward": -1.0631978923925089e-07,
"reward_std": 0.9958245873451232,
"rewards/rnd_reward/mean": -1.0631978462072311e-07,
"rewards/rnd_reward/std": 0.9958245933055878,
"step": 1460,
"step_time": 10.52764433112461
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.12750000283122062,
"completions/max_length": 384.0,
"completions/max_terminated_length": 347.3,
"completions/mean_length": 111.42625198364257,
"completions/mean_terminated_length": 71.7352523803711,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.8058162331581116,
"epoch": 0.961412688031393,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9467556476593018,
"kl": 0.07855136888101696,
"learning_rate": 2.598103335513407e-07,
"loss": 0.10095235109329223,
"num_tokens": 30671516.0,
"reward": -1.529604193084566e-07,
"reward_std": 0.9958255529403687,
"rewards/rnd_reward/mean": -1.5296041966372798e-07,
"rewards/rnd_reward/std": 0.9958255529403687,
"step": 1470,
"step_time": 10.889759617857635
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.11125000268220901,
"completions/max_length": 384.0,
"completions/max_terminated_length": 348.2,
"completions/mean_length": 105.6962516784668,
"completions/mean_terminated_length": 70.84101905822754,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.9369604349136353,
"epoch": 0.9679529103989536,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9438734650611877,
"kl": 0.0841084063053131,
"learning_rate": 2.581752779594506e-07,
"loss": 0.06790788173675537,
"num_tokens": 30852283.0,
"reward": -1.862645504502325e-08,
"reward_std": 0.9955705404281616,
"rewards/rnd_reward/mean": -1.8626451847580937e-08,
"rewards/rnd_reward/std": 0.9955705344676972,
"step": 1480,
"step_time": 10.544715784140863
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.11250000223517417,
"completions/max_length": 384.0,
"completions/max_terminated_length": 347.8,
"completions/mean_length": 111.51875228881836,
"completions/mean_terminated_length": 77.10519866943359,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.871400186419487,
"epoch": 0.9744931327665141,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.0710206031799316,
"kl": 0.07257421473041177,
"learning_rate": 2.5654022236756053e-07,
"loss": 0.10279649496078491,
"num_tokens": 31039368.0,
"reward": -3.039836773410798e-08,
"reward_std": 0.9956327259540558,
"rewards/rnd_reward/mean": -3.039836915519345e-08,
"rewards/rnd_reward/std": 0.9956327199935913,
"step": 1490,
"step_time": 11.004371847352013
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.11750000305473804,
"completions/max_length": 384.0,
"completions/max_terminated_length": 340.9,
"completions/mean_length": 110.3587516784668,
"completions/mean_terminated_length": 74.05159416198731,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.8570073008537293,
"epoch": 0.9810333551340745,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9674124717712402,
"kl": 0.07076722700148821,
"learning_rate": 2.5490516677567035e-07,
"loss": 0.09354864358901978,
"num_tokens": 31229300.0,
"reward": -7.458031374696361e-08,
"reward_std": 0.9957621216773986,
"rewards/rnd_reward/mean": -7.458031019424993e-08,
"rewards/rnd_reward/std": 0.9957621157169342,
"step": 1500,
"step_time": 10.672748254612088
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09250000156462193,
"completions/max_length": 384.0,
"completions/max_terminated_length": 366.3,
"completions/mean_length": 109.77875213623047,
"completions/mean_terminated_length": 81.85236892700195,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.8984388679265978,
"epoch": 0.9875735775016351,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.144245982170105,
"kl": 0.07264794735237956,
"learning_rate": 2.5327011118378023e-07,
"loss": 0.10363882780075073,
"num_tokens": 31410018.0,
"reward": 3.874301679474002e-09,
"reward_std": 0.9956199109554291,
"rewards/rnd_reward/mean": 3.874301679474002e-09,
"rewards/rnd_reward/std": 0.9956199288368225,
"step": 1510,
"step_time": 10.57373680251185
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.11250000223517417,
"completions/max_length": 384.0,
"completions/max_terminated_length": 352.4,
"completions/mean_length": 107.2912498474121,
"completions/mean_terminated_length": 71.87631187438964,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.835093083977699,
"epoch": 0.9941137998691956,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9730135798454285,
"kl": 0.06768459901213646,
"learning_rate": 2.516350555918901e-07,
"loss": 0.04406432807445526,
"num_tokens": 31584581.0,
"reward": -6.169080961626605e-08,
"reward_std": 0.9956278324127197,
"rewards/rnd_reward/mean": -6.169080748463785e-08,
"rewards/rnd_reward/std": 0.9956278324127197,
"step": 1520,
"step_time": 9.995071636443026
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.10000000149011612,
"completions/max_length": 384.0,
"completions/max_terminated_length": 356.7,
"completions/mean_length": 105.59500198364258,
"completions/mean_terminated_length": 74.4354133605957,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.818147099018097,
"epoch": 1.0006540222367561,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.0657150745391846,
"kl": 0.0751381286419928,
"learning_rate": 2.5e-07,
"loss": 0.14666662216186524,
"num_tokens": 31762082.0,
"reward": -7.718801313671974e-08,
"reward_std": 0.9956592619419098,
"rewards/rnd_reward/mean": -7.718801242617701e-08,
"rewards/rnd_reward/std": 0.9956592619419098,
"step": 1530,
"step_time": 10.494197542965413
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.08500000089406967,
"completions/max_length": 384.0,
"completions/max_terminated_length": 346.0,
"completions/mean_length": 100.37250137329102,
"completions/mean_terminated_length": 73.99925155639649,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.850705623626709,
"epoch": 1.0071942446043165,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8915496468544006,
"kl": 0.08450272670015693,
"learning_rate": 2.4836494440810987e-07,
"loss": 0.10265235900878907,
"num_tokens": 31935920.0,
"reward": -7.78958216152148e-08,
"reward_std": 0.9955736041069031,
"rewards/rnd_reward/mean": -7.789581664141565e-08,
"rewards/rnd_reward/std": 0.995573616027832,
"step": 1540,
"step_time": 10.282078485726379
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09625000357627869,
"completions/max_length": 384.0,
"completions/max_terminated_length": 339.5,
"completions/mean_length": 100.54375,
"completions/mean_terminated_length": 70.20137939453124,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.7832202523946763,
"epoch": 1.013734466971877,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9670330882072449,
"kl": 0.07673759749159217,
"learning_rate": 2.4672988881621975e-07,
"loss": 0.053245991468429565,
"num_tokens": 32105575.0,
"reward": -7.98702268323126e-08,
"reward_std": 0.9954982161521911,
"rewards/rnd_reward/mean": -7.987022256905619e-08,
"rewards/rnd_reward/std": 0.9954982340335846,
"step": 1550,
"step_time": 10.094654931733384
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09625000134110451,
"completions/max_length": 384.0,
"completions/max_terminated_length": 356.6,
"completions/mean_length": 105.8862518310547,
"completions/mean_terminated_length": 76.32862396240235,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.806140884757042,
"epoch": 1.0202746893394374,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9716780185699463,
"kl": 0.07651977436617016,
"learning_rate": 2.450948332243296e-07,
"loss": 0.07588104605674743,
"num_tokens": 32282264.0,
"reward": -1.2878328945475914e-07,
"reward_std": 0.9953814685344696,
"rewards/rnd_reward/mean": -1.2878328448096e-07,
"rewards/rnd_reward/std": 0.9953814685344696,
"step": 1560,
"step_time": 10.249592515034601
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09250000044703484,
"completions/max_length": 384.0,
"completions/max_terminated_length": 370.2,
"completions/mean_length": 101.42500152587891,
"completions/mean_terminated_length": 72.58811225891114,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.706285873055458,
"epoch": 1.026814911706998,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.2053457498550415,
"kl": 0.07990584680810571,
"learning_rate": 2.434597776324395e-07,
"loss": 0.08478202223777771,
"num_tokens": 32458489.0,
"reward": -1.197308309830447e-07,
"reward_std": 0.9958563566207885,
"rewards/rnd_reward/mean": -1.1973082814087377e-07,
"rewards/rnd_reward/std": 0.9958563566207885,
"step": 1570,
"step_time": 10.385891513410025
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.08750000223517418,
"completions/max_length": 384.0,
"completions/max_terminated_length": 351.3,
"completions/mean_length": 92.8375015258789,
"completions/mean_terminated_length": 65.02403182983399,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.832634910941124,
"epoch": 1.0333551340745586,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.2888636589050293,
"kl": 0.08601820273324848,
"learning_rate": 2.418247220405494e-07,
"loss": 0.024559563398361205,
"num_tokens": 32626059.0,
"reward": -1.628696985278566e-07,
"reward_std": 0.9954473912715912,
"rewards/rnd_reward/mean": -1.6286969000134378e-07,
"rewards/rnd_reward/std": 0.9954474031925201,
"step": 1580,
"step_time": 10.754275933234021
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09125000201165676,
"completions/max_length": 384.0,
"completions/max_terminated_length": 358.7,
"completions/mean_length": 99.23750076293945,
"completions/mean_terminated_length": 70.4692756652832,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.9075894236564634,
"epoch": 1.039895356442119,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.0340582132339478,
"kl": 0.0824001532047987,
"learning_rate": 2.4018966644865926e-07,
"loss": 0.04327432513236999,
"num_tokens": 32797839.0,
"reward": 6.616115513224941e-08,
"reward_std": 0.9953638434410095,
"rewards/rnd_reward/mean": 6.61611530006212e-08,
"rewards/rnd_reward/std": 0.9953638732433319,
"step": 1590,
"step_time": 10.39529764095787
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.1075000025331974,
"completions/max_length": 384.0,
"completions/max_terminated_length": 363.2,
"completions/mean_length": 105.15875091552735,
"completions/mean_terminated_length": 71.44011192321777,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 3.068614384531975,
"epoch": 1.0464355788096795,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9280741810798645,
"kl": 0.08188259806483984,
"learning_rate": 2.3855461085676914e-07,
"loss": 0.07503924369812012,
"num_tokens": 32970486.0,
"reward": 2.354383532576776e-08,
"reward_std": 0.9955627143383026,
"rewards/rnd_reward/mean": 2.3543834259953656e-08,
"rewards/rnd_reward/std": 0.9955627143383026,
"step": 1600,
"step_time": 10.102484514610842
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.12375000268220901,
"completions/max_length": 384.0,
"completions/max_terminated_length": 352.0,
"completions/mean_length": 108.67875137329102,
"completions/mean_terminated_length": 69.76623840332032,
"completions/min_length": 2.4,
"completions/min_terminated_length": 2.4,
"entropy": 2.7116469889879227,
"epoch": 1.05297580117724,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8361508846282959,
"kl": 0.07135752588510513,
"learning_rate": 2.36919555264879e-07,
"loss": 0.08789730072021484,
"num_tokens": 33155644.0,
"reward": -7.331371434382561e-08,
"reward_std": 0.9953298509120941,
"rewards/rnd_reward/mean": -7.331371363328287e-08,
"rewards/rnd_reward/std": 0.9953298389911651,
"step": 1610,
"step_time": 10.373006676253862
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.10250000208616257,
"completions/max_length": 384.0,
"completions/max_terminated_length": 362.6,
"completions/mean_length": 107.08625106811523,
"completions/mean_terminated_length": 75.19281349182128,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.767005005478859,
"epoch": 1.0595160235448005,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8207447528839111,
"kl": 0.07195558175444602,
"learning_rate": 2.352844996729889e-07,
"loss": 0.07792651653289795,
"num_tokens": 33333863.0,
"reward": -7.919967437430842e-08,
"reward_std": 0.9952020287513733,
"rewards/rnd_reward/mean": -7.919966940050927e-08,
"rewards/rnd_reward/std": 0.9952020525932312,
"step": 1620,
"step_time": 10.309986177436077
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.12875000163912773,
"completions/max_length": 384.0,
"completions/max_terminated_length": 355.7,
"completions/mean_length": 117.29250259399414,
"completions/mean_terminated_length": 78.07871360778809,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.8041924834251404,
"epoch": 1.066056245912361,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.7970576882362366,
"kl": 0.07171947872266174,
"learning_rate": 2.3364944408109875e-07,
"loss": 0.07321782112121582,
"num_tokens": 33529112.0,
"reward": -7.54743808784042e-08,
"reward_std": 0.9954242706298828,
"rewards/rnd_reward/mean": -7.54743808784042e-08,
"rewards/rnd_reward/std": 0.9954242765903473,
"step": 1630,
"step_time": 11.020952883246355
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.10375000089406967,
"completions/max_length": 384.0,
"completions/max_terminated_length": 367.8,
"completions/mean_length": 109.85750122070313,
"completions/mean_terminated_length": 78.08732643127442,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.7948483556509016,
"epoch": 1.0725964682799216,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.0839210748672485,
"kl": 0.07083125235512852,
"learning_rate": 2.3201438848920862e-07,
"loss": 0.07660450935363769,
"num_tokens": 33711328.0,
"reward": -4.068016963287846e-08,
"reward_std": 0.9954971015453339,
"rewards/rnd_reward/mean": -4.068016963287846e-08,
"rewards/rnd_reward/std": 0.9954971194267273,
"step": 1640,
"step_time": 10.339221865893341
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09125000033527612,
"completions/max_length": 384.0,
"completions/max_terminated_length": 355.3,
"completions/mean_length": 100.59000167846679,
"completions/mean_terminated_length": 72.11472930908204,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.7371392965316774,
"epoch": 1.079136690647482,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9994949698448181,
"kl": 0.08674983121454716,
"learning_rate": 2.303793328973185e-07,
"loss": 0.14760398864746094,
"num_tokens": 33894090.0,
"reward": -4.775822315394862e-08,
"reward_std": 0.9951541960239411,
"rewards/rnd_reward/mean": -4.7758222088134514e-08,
"rewards/rnd_reward/std": 0.9951542019844055,
"step": 1650,
"step_time": 11.004630160145462
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09750000163912773,
"completions/max_length": 384.0,
"completions/max_terminated_length": 353.6,
"completions/mean_length": 103.39125213623046,
"completions/mean_terminated_length": 73.03910675048829,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.719507083296776,
"epoch": 1.0856769130150425,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.0636616945266724,
"kl": 0.07622466497123241,
"learning_rate": 2.2874427730542835e-07,
"loss": 0.11726703643798828,
"num_tokens": 34065638.0,
"reward": -6.273389079325397e-08,
"reward_std": 0.9954430878162384,
"rewards/rnd_reward/mean": -6.273388937216851e-08,
"rewards/rnd_reward/std": 0.9954430878162384,
"step": 1660,
"step_time": 10.205868328805082
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.11125000230967999,
"completions/max_length": 384.0,
"completions/max_terminated_length": 337.9,
"completions/mean_length": 103.16000251770019,
"completions/mean_terminated_length": 68.0124439239502,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.555436444282532,
"epoch": 1.0922171353826031,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.4149969816207886,
"kl": 0.08045686418190598,
"learning_rate": 2.2710922171353826e-07,
"loss": 0.18388016223907472,
"num_tokens": 34243911.0,
"reward": -6.049871625890546e-08,
"reward_std": 0.995277750492096,
"rewards/rnd_reward/mean": -6.049871181801336e-08,
"rewards/rnd_reward/std": 0.9952777206897736,
"step": 1670,
"step_time": 10.405935624614358
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09250000230967999,
"completions/max_length": 384.0,
"completions/max_terminated_length": 349.5,
"completions/mean_length": 98.8687515258789,
"completions/mean_terminated_length": 69.8191291809082,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.679945033788681,
"epoch": 1.0987573577501635,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.1418001651763916,
"kl": 0.07972979731857777,
"learning_rate": 2.2547416612164814e-07,
"loss": 0.08271819949150086,
"num_tokens": 34422086.0,
"reward": -2.2441148974650105e-07,
"reward_std": 0.9950048804283143,
"rewards/rnd_reward/mean": -2.2441148654905875e-07,
"rewards/rnd_reward/std": 0.9950048506259919,
"step": 1680,
"step_time": 10.670063342596404
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.08875000141561032,
"completions/max_length": 384.0,
"completions/max_terminated_length": 348.9,
"completions/mean_length": 95.35000228881836,
"completions/mean_terminated_length": 67.31705818176269,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.6938273787498472,
"epoch": 1.105297580117724,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.0555897951126099,
"kl": 0.0881148618645966,
"learning_rate": 2.2383911052975802e-07,
"loss": 0.057183104753494265,
"num_tokens": 34592661.0,
"reward": -5.260109574578564e-08,
"reward_std": 0.9953494489192962,
"rewards/rnd_reward/mean": -5.260109787741385e-08,
"rewards/rnd_reward/std": 0.9953494608402252,
"step": 1690,
"step_time": 10.320013692067004
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.09500000197440386,
"completions/max_length": 384.0,
"completions/max_terminated_length": 351.8,
"completions/mean_length": 96.38625259399414,
"completions/mean_terminated_length": 66.25210113525391,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.8528232872486115,
"epoch": 1.1118378024852844,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.1842913627624512,
"kl": 0.09175272611901164,
"learning_rate": 2.2220405493786787e-07,
"loss": 0.11493265628814697,
"num_tokens": 34767800.0,
"reward": -3.5762789174498266e-08,
"reward_std": 0.9952230334281922,
"rewards/rnd_reward/mean": -3.5762784911241854e-08,
"rewards/rnd_reward/std": 0.9952230513095855,
"step": 1700,
"step_time": 11.20805806266144
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.06625000089406967,
"completions/max_length": 384.0,
"completions/max_terminated_length": 349.7,
"completions/mean_length": 88.85,
"completions/mean_terminated_length": 67.81800193786621,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.7997246354818346,
"epoch": 1.118378024852845,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.3541666269302368,
"kl": 0.0883994361385703,
"learning_rate": 2.2056899934597775e-07,
"loss": 0.10645198822021484,
"num_tokens": 34940290.0,
"reward": -9.074807572062582e-08,
"reward_std": 0.9952512562274933,
"rewards/rnd_reward/mean": -9.074807181264077e-08,
"rewards/rnd_reward/std": 0.9952512562274933,
"step": 1710,
"step_time": 11.268909463495948
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.08375000078231096,
"completions/max_length": 384.0,
"completions/max_terminated_length": 341.1,
"completions/mean_length": 92.34500198364258,
"completions/mean_terminated_length": 65.63225555419922,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.5688311845064162,
"epoch": 1.1249182472204056,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.086549997329712,
"kl": 0.08725747494027018,
"learning_rate": 2.1893394375408762e-07,
"loss": 0.09306548833847046,
"num_tokens": 35107316.0,
"reward": -1.1682510461241691e-07,
"reward_std": 0.9951877236366272,
"rewards/rnd_reward/mean": -1.1682510461241691e-07,
"rewards/rnd_reward/std": 0.9951877236366272,
"step": 1720,
"step_time": 10.925348741887138
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.08875000104308128,
"completions/max_length": 384.0,
"completions/max_terminated_length": 349.1,
"completions/mean_length": 97.91875076293945,
"completions/mean_terminated_length": 70.1454303741455,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.690840741991997,
"epoch": 1.131458469587966,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.0729917287826538,
"kl": 0.08789877109229564,
"learning_rate": 2.172988881621975e-07,
"loss": 0.10979057550430298,
"num_tokens": 35281881.0,
"reward": -1.907348732288483e-08,
"reward_std": 0.995078319311142,
"rewards/rnd_reward/mean": -1.907348732288483e-08,
"rewards/rnd_reward/std": 0.9950783312320709,
"step": 1730,
"step_time": 10.632008279883303
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.08750000223517418,
"completions/max_length": 384.0,
"completions/max_terminated_length": 355.8,
"completions/mean_length": 96.91125183105468,
"completions/mean_terminated_length": 69.48632164001465,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.681867742538452,
"epoch": 1.1379986919555265,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.0131276845932007,
"kl": 0.08944809632375836,
"learning_rate": 2.1566383257030738e-07,
"loss": 0.10518434047698974,
"num_tokens": 35461970.0,
"reward": 2.6673077968553116e-08,
"reward_std": 0.9949174404144288,
"rewards/rnd_reward/mean": 2.6673081521266795e-08,
"rewards/rnd_reward/std": 0.9949174463748932,
"step": 1740,
"step_time": 11.253412028984167
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0787500012665987,
"completions/max_length": 384.0,
"completions/max_terminated_length": 338.7,
"completions/mean_length": 84.14750061035156,
"completions/mean_terminated_length": 58.63842086791992,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.567233794927597,
"epoch": 1.144538914323087,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.1535600423812866,
"kl": 0.09054145393893123,
"learning_rate": 2.1402877697841726e-07,
"loss": 0.10667283535003662,
"num_tokens": 35622023.0,
"reward": -5.97536580215774e-08,
"reward_std": 0.9952092111110687,
"rewards/rnd_reward/mean": -5.975365517940645e-08,
"rewards/rnd_reward/std": 0.9952091991901397,
"step": 1750,
"step_time": 10.509924068162217
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.07000000029802322,
"completions/max_length": 384.0,
"completions/max_terminated_length": 331.8,
"completions/mean_length": 86.03000106811524,
"completions/mean_terminated_length": 63.5996639251709,
"completions/min_length": 2.4,
"completions/min_terminated_length": 2.4,
"entropy": 2.5665056079626085,
"epoch": 1.1510791366906474,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.1171236038208008,
"kl": 0.08619214734062552,
"learning_rate": 2.1239372138652714e-07,
"loss": 0.07680293321609497,
"num_tokens": 35780972.0,
"reward": -4.529952919440916e-08,
"reward_std": 0.9950806498527527,
"rewards/rnd_reward/mean": -4.529952919440916e-08,
"rewards/rnd_reward/std": 0.9950806617736816,
"step": 1760,
"step_time": 10.277455605636352
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.06875000074505806,
"completions/max_length": 384.0,
"completions/max_terminated_length": 345.5,
"completions/mean_length": 89.81625213623047,
"completions/mean_terminated_length": 68.19218559265137,
"completions/min_length": 2.2,
"completions/min_terminated_length": 2.2,
"entropy": 2.563973236083984,
"epoch": 1.157619359058208,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.1521550416946411,
"kl": 0.0870651968754828,
"learning_rate": 2.10758665794637e-07,
"loss": 0.1115831971168518,
"num_tokens": 35948090.0,
"reward": 1.3858081859829951e-08,
"reward_std": 0.9948914706707,
"rewards/rnd_reward/mean": 1.38580813269229e-08,
"rewards/rnd_reward/std": 0.9948914825916291,
"step": 1770,
"step_time": 10.7476754687028
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.07750000115483999,
"completions/max_length": 384.0,
"completions/max_terminated_length": 346.2,
"completions/mean_length": 85.87875022888184,
"completions/mean_terminated_length": 60.88022689819336,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.646288427710533,
"epoch": 1.1641595814257686,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9658340811729431,
"kl": 0.09920121757313609,
"learning_rate": 2.091236102027469e-07,
"loss": 0.08330439329147339,
"num_tokens": 36114108.0,
"reward": -7.227063285597523e-08,
"reward_std": 0.9949531733989716,
"rewards/rnd_reward/mean": -7.227063250070387e-08,
"rewards/rnd_reward/std": 0.994953167438507,
"step": 1780,
"step_time": 11.134026414644904
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.06250000018626452,
"completions/max_length": 384.0,
"completions/max_terminated_length": 355.1,
"completions/mean_length": 84.51375122070313,
"completions/mean_terminated_length": 64.53133659362793,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.704926884174347,
"epoch": 1.170699803793329,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.2755341529846191,
"kl": 0.10716334227472543,
"learning_rate": 2.0748855461085677e-07,
"loss": 0.13019881248474122,
"num_tokens": 36271504.0,
"reward": -1.305341750246214e-07,
"reward_std": 0.9949664771556854,
"rewards/rnd_reward/mean": -1.305341714719077e-07,
"rewards/rnd_reward/std": 0.9949664771556854,
"step": 1790,
"step_time": 10.397847580048255
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.08625000175088644,
"completions/max_length": 384.0,
"completions/max_terminated_length": 326.3,
"completions/mean_length": 87.63249969482422,
"completions/mean_terminated_length": 59.75216636657715,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.593474331498146,
"epoch": 1.1772400261608895,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.155380129814148,
"kl": 0.09049692396074534,
"learning_rate": 2.0585349901896665e-07,
"loss": 0.12421066761016845,
"num_tokens": 36439860.0,
"reward": 1.50501714202278e-08,
"reward_std": 0.9950001060962677,
"rewards/rnd_reward/mean": 1.505017284131327e-08,
"rewards/rnd_reward/std": 0.9950001001358032,
"step": 1800,
"step_time": 10.84767303718254
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.06250000158324838,
"completions/max_length": 384.0,
"completions/max_terminated_length": 355.8,
"completions/mean_length": 91.63375244140624,
"completions/mean_terminated_length": 72.32135162353515,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.5027853429317473,
"epoch": 1.1837802485284499,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.0669174194335938,
"kl": 0.08057954236865043,
"learning_rate": 2.042184434270765e-07,
"loss": 0.09477431774139404,
"num_tokens": 36602592.0,
"reward": -1.505017284131327e-07,
"reward_std": 0.994869589805603,
"rewards/rnd_reward/mean": -1.5050172876840407e-07,
"rewards/rnd_reward/std": 0.9948696017265319,
"step": 1810,
"step_time": 10.333302733465098
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.07625000216066838,
"completions/max_length": 384.0,
"completions/max_terminated_length": 357.4,
"completions/mean_length": 92.49750137329102,
"completions/mean_terminated_length": 68.43786659240723,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.487256759405136,
"epoch": 1.1903204708960105,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.864107608795166,
"kl": 0.07850290769711137,
"learning_rate": 2.0258338783518638e-07,
"loss": 0.09753584861755371,
"num_tokens": 36768390.0,
"reward": -1.2189149920516229e-07,
"reward_std": 0.9947888195514679,
"rewards/rnd_reward/mean": -1.2189150062624775e-07,
"rewards/rnd_reward/std": 0.9947888076305389,
"step": 1820,
"step_time": 10.366325606242754
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.07500000111758709,
"completions/max_length": 384.0,
"completions/max_terminated_length": 327.2,
"completions/mean_length": 87.97500152587891,
"completions/mean_terminated_length": 64.1132999420166,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.5121534377336503,
"epoch": 1.196860693263571,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.8458340167999268,
"kl": 0.08256649980321526,
"learning_rate": 2.0094833224329626e-07,
"loss": 0.06835322976112365,
"num_tokens": 36933960.0,
"reward": -1.531839394885992e-07,
"reward_std": 0.9951117813587189,
"rewards/rnd_reward/mean": -1.5318393664642826e-07,
"rewards/rnd_reward/std": 0.9951117634773254,
"step": 1830,
"step_time": 11.365050849993713
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.07375000137835741,
"completions/max_length": 384.0,
"completions/max_terminated_length": 363.0,
"completions/mean_length": 94.00125045776367,
"completions/mean_terminated_length": 71.0650505065918,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.4178095728158953,
"epoch": 1.2034009156311314,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.1388354301452637,
"kl": 0.0803013733588159,
"learning_rate": 1.9931327665140614e-07,
"loss": 0.12810555696487427,
"num_tokens": 37099636.0,
"reward": -4.909932687091611e-08,
"reward_std": 0.9949400007724762,
"rewards/rnd_reward/mean": -4.909932758145885e-08,
"rewards/rnd_reward/std": 0.9949400067329407,
"step": 1840,
"step_time": 10.505639822757804
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.06875000093132258,
"completions/max_length": 384.0,
"completions/max_terminated_length": 341.5,
"completions/mean_length": 91.8550018310547,
"completions/mean_terminated_length": 70.10141868591309,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.4157417356967925,
"epoch": 1.209941137998692,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.0894784927368164,
"kl": 0.08061180328950286,
"learning_rate": 1.9767822105951601e-07,
"loss": 0.09716765880584717,
"num_tokens": 37269225.0,
"reward": -4.3064359189770585e-08,
"reward_std": 0.9946663022041321,
"rewards/rnd_reward/mean": -4.306435528178554e-08,
"rewards/rnd_reward/std": 0.9946663022041321,
"step": 1850,
"step_time": 11.422960984380916
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.08750000074505807,
"completions/max_length": 384.0,
"completions/max_terminated_length": 350.2,
"completions/mean_length": 93.6987518310547,
"completions/mean_terminated_length": 65.78671073913574,
"completions/min_length": 2.2,
"completions/min_terminated_length": 2.2,
"entropy": 2.5770903497934343,
"epoch": 1.2164813603662525,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9160525798797607,
"kl": 0.08047085385769606,
"learning_rate": 1.960431654676259e-07,
"loss": 0.05968161225318909,
"num_tokens": 37440344.0,
"reward": 2.6524067830990816e-08,
"reward_std": 0.9945189356803894,
"rewards/rnd_reward/mean": 2.65240635677344e-08,
"rewards/rnd_reward/std": 0.9945189356803894,
"step": 1860,
"step_time": 10.725451961229556
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.08625000063329935,
"completions/max_length": 384.0,
"completions/max_terminated_length": 335.1,
"completions/mean_length": 95.60625076293945,
"completions/mean_terminated_length": 68.25989952087403,
"completions/min_length": 2.3,
"completions/min_terminated_length": 2.3,
"entropy": 2.4541951596736906,
"epoch": 1.223021582733813,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.999136209487915,
"kl": 0.07508358927443623,
"learning_rate": 1.9440810987573577e-07,
"loss": 0.13460888862609863,
"num_tokens": 37611004.0,
"reward": -1.23679642705099e-07,
"reward_std": 0.9949141681194306,
"rewards/rnd_reward/mean": -1.236796327575007e-07,
"rewards/rnd_reward/std": 0.9949141621589661,
"step": 1870,
"step_time": 10.482190745323896
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.07625000178813934,
"completions/max_length": 384.0,
"completions/max_terminated_length": 328.5,
"completions/mean_length": 89.38125152587891,
"completions/mean_terminated_length": 65.09402084350586,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.567149502038956,
"epoch": 1.2295618051013735,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.2019909620285034,
"kl": 0.08660045554861426,
"learning_rate": 1.9277305428384562e-07,
"loss": 0.20026679039001466,
"num_tokens": 37786329.0,
"reward": -8.866190821521514e-08,
"reward_std": 0.9948227822780609,
"rewards/rnd_reward/mean": -8.866190999157198e-08,
"rewards/rnd_reward/std": 0.9948227941989899,
"step": 1880,
"step_time": 11.85091487034224
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.06375000011175871,
"completions/max_length": 384.0,
"completions/max_terminated_length": 345.9,
"completions/mean_length": 89.55750122070313,
"completions/mean_terminated_length": 69.49982872009278,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.8322397589683534,
"epoch": 1.236102027468934,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.1517750024795532,
"kl": 0.09399051470682025,
"learning_rate": 1.9113799869195553e-07,
"loss": 0.17799322605133056,
"num_tokens": 37947720.0,
"reward": -1.0848045732814171e-07,
"reward_std": 0.9948072552680969,
"rewards/rnd_reward/mean": -1.0848045448597076e-07,
"rewards/rnd_reward/std": 0.9948072493076324,
"step": 1890,
"step_time": 10.414563778741286
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.06625000014901161,
"completions/max_length": 384.0,
"completions/max_terminated_length": 347.4,
"completions/mean_length": 86.39500045776367,
"completions/mean_terminated_length": 65.30856475830078,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.7242774665355682,
"epoch": 1.2426422498364944,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.3484736680984497,
"kl": 0.09410189539194107,
"learning_rate": 1.895029431000654e-07,
"loss": 0.0743591845035553,
"num_tokens": 38111876.0,
"reward": -6.467104114449285e-08,
"reward_std": 0.9947612941265106,
"rewards/rnd_reward/mean": -6.467103901286464e-08,
"rewards/rnd_reward/std": 0.9947613418102265,
"step": 1900,
"step_time": 10.527994227781893
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.06625000108033419,
"completions/max_length": 384.0,
"completions/max_terminated_length": 338.0,
"completions/mean_length": 84.77500114440917,
"completions/mean_terminated_length": 63.568695068359375,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.954171320796013,
"epoch": 1.249182472204055,
"frac_reward_zero_std": 0.0,
"grad_norm": 0.9760697484016418,
"kl": 0.10021719867363572,
"learning_rate": 1.8786788750817528e-07,
"loss": 0.02971080243587494,
"num_tokens": 38279376.0,
"reward": -1.5616416861874428e-07,
"reward_std": 0.9947503983974457,
"rewards/rnd_reward/mean": -1.5616416790820153e-07,
"rewards/rnd_reward/std": 0.9947503864765167,
"step": 1910,
"step_time": 11.339212405891157
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.08000000193715096,
"completions/max_length": 384.0,
"completions/max_terminated_length": 345.5,
"completions/mean_length": 88.47500076293946,
"completions/mean_terminated_length": 62.80438385009766,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.846560522913933,
"epoch": 1.2557226945716153,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.2370659112930298,
"kl": 0.09515976402908563,
"learning_rate": 1.8623283191628514e-07,
"loss": 0.10529450178146363,
"num_tokens": 38449416.0,
"reward": -3.65078278719011e-09,
"reward_std": 0.9944445490837097,
"rewards/rnd_reward/mean": -3.6507834977328456e-09,
"rewards/rnd_reward/std": 0.9944445431232453,
"step": 1920,
"step_time": 10.698268468677998
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.07625000160187483,
"completions/max_length": 384.0,
"completions/max_terminated_length": 344.8,
"completions/mean_length": 87.2925018310547,
"completions/mean_terminated_length": 62.797546768188475,
"completions/min_length": 2.2,
"completions/min_terminated_length": 2.2,
"entropy": 2.9005165725946425,
"epoch": 1.262262916939176,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.3234843015670776,
"kl": 0.10119491461664439,
"learning_rate": 1.8459777632439501e-07,
"loss": 0.014341549575328827,
"num_tokens": 38616700.0,
"reward": -4.380941049930698e-08,
"reward_std": 0.9944035768508911,
"rewards/rnd_reward/mean": -4.3809414407292026e-08,
"rewards/rnd_reward/std": 0.994403600692749,
"step": 1930,
"step_time": 11.375351972319185
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.08750000111758709,
"completions/max_length": 384.0,
"completions/max_terminated_length": 334.7,
"completions/mean_length": 96.3550018310547,
"completions/mean_terminated_length": 68.85774459838868,
"completions/min_length": 2.4,
"completions/min_terminated_length": 2.4,
"entropy": 2.9528768658638,
"epoch": 1.2688031393067365,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.1305567026138306,
"kl": 0.09469359386712313,
"learning_rate": 1.829627207325049e-07,
"loss": 0.0362447589635849,
"num_tokens": 38788569.0,
"reward": -5.45382517813664e-08,
"reward_std": 0.9948436379432678,
"rewards/rnd_reward/mean": -5.4538249649738193e-08,
"rewards/rnd_reward/std": 0.9948436439037323,
"step": 1940,
"step_time": 10.781422062939964
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.10000000298023223,
"completions/max_length": 384.0,
"completions/max_terminated_length": 342.1,
"completions/mean_length": 93.3875015258789,
"completions/mean_terminated_length": 61.06837921142578,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.7504739254713058,
"epoch": 1.2753433616742968,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.153893232345581,
"kl": 0.08643652414903044,
"learning_rate": 1.8132766514061477e-07,
"loss": 0.10290535688400268,
"num_tokens": 38954489.0,
"reward": -1.2144446595030445e-07,
"reward_std": 0.9947971343994141,
"rewards/rnd_reward/mean": -1.2144446346340487e-07,
"rewards/rnd_reward/std": 0.994797146320343,
"step": 1950,
"step_time": 10.527900594798847
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0950000025331974,
"completions/max_length": 384.0,
"completions/max_terminated_length": 329.3,
"completions/mean_length": 100.58875122070313,
"completions/mean_terminated_length": 70.79409141540528,
"completions/min_length": 2.4,
"completions/min_terminated_length": 2.4,
"entropy": 2.7141620010137557,
"epoch": 1.2818835840418574,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.0054807662963867,
"kl": 0.08213893175125123,
"learning_rate": 1.7969260954872465e-07,
"loss": 0.07241884469985962,
"num_tokens": 39126695.0,
"reward": -1.028180136586343e-08,
"reward_std": 0.9948840856552124,
"rewards/rnd_reward/mean": -1.0281802786948902e-08,
"rewards/rnd_reward/std": 0.9948840975761414,
"step": 1960,
"step_time": 10.461326698656194
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.06750000081956387,
"completions/max_length": 383.2,
"completions/max_terminated_length": 350.0,
"completions/mean_length": 88.08000106811524,
"completions/mean_terminated_length": 66.5671230316162,
"completions/min_length": 2.3,
"completions/min_terminated_length": 2.3,
"entropy": 2.7966362595558167,
"epoch": 1.2884238064094178,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.1359318494796753,
"kl": 0.09009936284273863,
"learning_rate": 1.7805755395683453e-07,
"loss": 0.12549155950546265,
"num_tokens": 39294724.0,
"reward": 4.082918199088681e-08,
"reward_std": 0.9944938480854034,
"rewards/rnd_reward/mean": 4.0829182701429545e-08,
"rewards/rnd_reward/std": 0.9944938659667969,
"step": 1970,
"step_time": 10.983743568998761
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.07625000178813934,
"completions/max_length": 384.0,
"completions/max_terminated_length": 340.4,
"completions/mean_length": 95.8025016784668,
"completions/mean_terminated_length": 71.99091606140136,
"completions/min_length": 2.0,
"completions/min_terminated_length": 2.0,
"entropy": 2.7956566989421843,
"epoch": 1.2949640287769784,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.3064550161361694,
"kl": 0.08204388581216335,
"learning_rate": 1.764224983649444e-07,
"loss": 0.06261605024337769,
"num_tokens": 39465981.0,
"reward": -7.681548792604076e-08,
"reward_std": 0.9945617258548737,
"rewards/rnd_reward/mean": -7.681548792604076e-08,
"rewards/rnd_reward/std": 0.9945617258548737,
"step": 1980,
"step_time": 10.856887764390558
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.10500000119209289,
"completions/max_length": 384.0,
"completions/max_terminated_length": 336.7,
"completions/mean_length": 99.39125137329101,
"completions/mean_terminated_length": 66.14606056213378,
"completions/min_length": 2.1,
"completions/min_terminated_length": 2.1,
"entropy": 2.809012603759766,
"epoch": 1.301504251144539,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.0052663087844849,
"kl": 0.08238375652581453,
"learning_rate": 1.7478744277305426e-07,
"loss": 0.04824514389038086,
"num_tokens": 39642064.0,
"reward": -2.3305416743113483e-07,
"reward_std": 0.9944523334503174,
"rewards/rnd_reward/mean": -2.3305416174679294e-07,
"rewards/rnd_reward/std": 0.9944523453712464,
"step": 1990,
"step_time": 10.668248945730738
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.06875000093132258,
"completions/max_length": 384.0,
"completions/max_terminated_length": 350.9,
"completions/mean_length": 94.09000167846679,
"completions/mean_terminated_length": 72.67349357604981,
"completions/min_length": 2.4,
"completions/min_terminated_length": 2.4,
"entropy": 2.782809841632843,
"epoch": 1.3080444735120995,
"frac_reward_zero_std": 0.0,
"grad_norm": 1.088478684425354,
"kl": 0.08519633868709206,
"learning_rate": 1.7315238718116416e-07,
"loss": 0.12635375261306764,
"num_tokens": 39807401.0,
"reward": -9.67830406706227e-08,
"reward_std": 0.9944651544094085,
"rewards/rnd_reward/mean": -9.67830406706227e-08,
"rewards/rnd_reward/std": 0.9944651544094085,
"step": 2000,
"step_time": 10.477193391043693
}
],
"logging_steps": 10,
"max_steps": 3058,
"num_input_tokens_seen": 39807401,
"num_train_epochs": 2,
"save_steps": 1000,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 0.0,
"train_batch_size": 20,
"trial_name": null,
"trial_params": null
}