1412 lines
49 KiB
JSON
1412 lines
49 KiB
JSON
{
|
|
"best_global_step": null,
|
|
"best_metric": null,
|
|
"best_model_checkpoint": null,
|
|
"epoch": 0.05,
|
|
"eval_steps": 500,
|
|
"global_step": 500,
|
|
"is_hyper_param_search": false,
|
|
"is_local_process_zero": true,
|
|
"is_world_process_zero": true,
|
|
"log_history": [
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.14998279511928558,
|
|
"epoch": 0.0001,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 2.8779168128967285,
|
|
"learning_rate": 1e-06,
|
|
"loss": -1.043081283569336e-07,
|
|
"num_tokens": 2568.0,
|
|
"reward": -0.09724999964237213,
|
|
"reward_std": 0.018171798437833786,
|
|
"rewards/_reward/mean": -0.09724999964237213,
|
|
"rewards/_reward/std": 0.018171798437833786,
|
|
"step": 1,
|
|
"step_time": 12.08820823195856
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.3047940664821201,
|
|
"epoch": 0.001,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 2.6030378341674805,
|
|
"learning_rate": 9.819999999999999e-07,
|
|
"loss": -8.27842288547092e-08,
|
|
"num_tokens": 27464.0,
|
|
"reward": -0.16016666839520136,
|
|
"reward_std": 0.06777944886643025,
|
|
"rewards/_reward/mean": -0.16016666839520136,
|
|
"rewards/_reward/std": 0.06777944886643025,
|
|
"step": 10,
|
|
"step_time": 11.564457062001262
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.2606607608497143,
|
|
"epoch": 0.002,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 3.2521839141845703,
|
|
"learning_rate": 9.619999999999999e-07,
|
|
"loss": 2.0712614059448243e-07,
|
|
"num_tokens": 55576.0,
|
|
"reward": -0.01043749824166298,
|
|
"reward_std": 0.0968616530764848,
|
|
"rewards/_reward/mean": -0.01043749824166298,
|
|
"rewards/_reward/std": 0.0968616530764848,
|
|
"step": 20,
|
|
"step_time": 11.661010127793997
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.24913175068795682,
|
|
"epoch": 0.003,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 1.774067759513855,
|
|
"learning_rate": 9.419999999999999e-07,
|
|
"loss": -3.3527612686157227e-07,
|
|
"num_tokens": 83448.0,
|
|
"reward": -0.13608750477433204,
|
|
"reward_std": 0.012762961606495083,
|
|
"rewards/_reward/mean": -0.13608750477433204,
|
|
"rewards/_reward/std": 0.012762961606495083,
|
|
"step": 30,
|
|
"step_time": 11.669710392298294
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.95,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 24.8,
|
|
"completions/mean_length": 254.8625,
|
|
"completions/mean_terminated_length": 23.325,
|
|
"completions/min_length": 252.0,
|
|
"completions/min_terminated_length": 21.6,
|
|
"entropy": 0.23657004088163375,
|
|
"epoch": 0.004,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 2.8075008392333984,
|
|
"learning_rate": 9.22e-07,
|
|
"loss": 0.002854938805103302,
|
|
"num_tokens": 111597.0,
|
|
"reward": -0.07128750085830689,
|
|
"reward_std": 0.08108895737677813,
|
|
"rewards/_reward/mean": -0.07128750085830689,
|
|
"rewards/_reward/std": 0.08108895737677813,
|
|
"step": 40,
|
|
"step_time": 11.579852976597612
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.22997551448643208,
|
|
"epoch": 0.005,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 3.7606520652770996,
|
|
"learning_rate": 9.02e-07,
|
|
"loss": 5.960464477539063e-09,
|
|
"num_tokens": 140101.0,
|
|
"reward": -0.10473749935626983,
|
|
"reward_std": 0.06175762424245477,
|
|
"rewards/_reward/mean": -0.10473749935626983,
|
|
"rewards/_reward/std": 0.06175762424245477,
|
|
"step": 50,
|
|
"step_time": 11.542670411110157
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.21263145469129086,
|
|
"epoch": 0.006,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 3.328623056411743,
|
|
"learning_rate": 8.82e-07,
|
|
"loss": -6.005167961120605e-07,
|
|
"num_tokens": 169069.0,
|
|
"reward": -0.11228750124573708,
|
|
"reward_std": 0.05499577845912427,
|
|
"rewards/_reward/mean": -0.11228750124573708,
|
|
"rewards/_reward/std": 0.05499577845912427,
|
|
"step": 60,
|
|
"step_time": 11.547676370796399
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.21824492514133453,
|
|
"epoch": 0.007,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 2.910665512084961,
|
|
"learning_rate": 8.62e-07,
|
|
"loss": 4.32133674621582e-08,
|
|
"num_tokens": 197517.0,
|
|
"reward": -0.1204499989748001,
|
|
"reward_std": 0.07779224249534308,
|
|
"rewards/_reward/mean": -0.1204499989748001,
|
|
"rewards/_reward/std": 0.07779224249534308,
|
|
"step": 70,
|
|
"step_time": 11.568741847097408
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.2399759404361248,
|
|
"epoch": 0.008,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 2.1106014251708984,
|
|
"learning_rate": 8.419999999999999e-07,
|
|
"loss": -3.3527612686157227e-08,
|
|
"num_tokens": 224933.0,
|
|
"reward": -0.010837499797344208,
|
|
"reward_std": 0.18366129244677723,
|
|
"rewards/_reward/mean": -0.010837499797344208,
|
|
"rewards/_reward/std": 0.18366129244677723,
|
|
"step": 80,
|
|
"step_time": 11.498653757106513
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.23416975382715463,
|
|
"epoch": 0.009,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 2.950016975402832,
|
|
"learning_rate": 8.219999999999999e-07,
|
|
"loss": -1.6205012798309326e-07,
|
|
"num_tokens": 252845.0,
|
|
"reward": -0.057250002026557924,
|
|
"reward_std": 0.11056184868793935,
|
|
"rewards/_reward/mean": -0.057250002026557924,
|
|
"rewards/_reward/std": 0.11056184868793935,
|
|
"step": 90,
|
|
"step_time": 11.616500220299349
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.9875,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 18.2,
|
|
"completions/mean_length": 255.075,
|
|
"completions/mean_terminated_length": 18.2,
|
|
"completions/min_length": 248.6,
|
|
"completions/min_terminated_length": 18.2,
|
|
"entropy": 0.2641608176752925,
|
|
"epoch": 0.01,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 2.9509315490722656,
|
|
"learning_rate": 8.02e-07,
|
|
"loss": 0.0012664616107940674,
|
|
"num_tokens": 280931.0,
|
|
"reward": -0.05811250358819962,
|
|
"reward_std": 0.088380962703377,
|
|
"rewards/_reward/mean": -0.05811250358819962,
|
|
"rewards/_reward/std": 0.088380962703377,
|
|
"step": 100,
|
|
"step_time": 11.67648275300453
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.23412359468638896,
|
|
"epoch": 0.011,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 4.738001346588135,
|
|
"learning_rate": 7.82e-07,
|
|
"loss": -9.760260581970215e-08,
|
|
"num_tokens": 308667.0,
|
|
"reward": -0.14102500230073928,
|
|
"reward_std": 0.012726937094703317,
|
|
"rewards/_reward/mean": -0.14102500230073928,
|
|
"rewards/_reward/std": 0.012726937094703317,
|
|
"step": 110,
|
|
"step_time": 11.954438662400936
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.19781138971447945,
|
|
"epoch": 0.012,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 2.452828884124756,
|
|
"learning_rate": 7.62e-07,
|
|
"loss": -3.3527612686157227e-08,
|
|
"num_tokens": 336835.0,
|
|
"reward": -0.11708749905228615,
|
|
"reward_std": 0.07922702981159091,
|
|
"rewards/_reward/mean": -0.11708749905228615,
|
|
"rewards/_reward/std": 0.07922702981159091,
|
|
"step": 120,
|
|
"step_time": 11.744920498199644
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.2342931807041168,
|
|
"epoch": 0.013,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 2.797684907913208,
|
|
"learning_rate": 7.42e-07,
|
|
"loss": 7.674098014831543e-08,
|
|
"num_tokens": 366091.0,
|
|
"reward": -0.11588749885559083,
|
|
"reward_std": 0.05765116480179131,
|
|
"rewards/_reward/mean": -0.11588749885559083,
|
|
"rewards/_reward/std": 0.05765116480179131,
|
|
"step": 130,
|
|
"step_time": 11.506405735507723
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.2076788205653429,
|
|
"epoch": 0.014,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 2.9772956371307373,
|
|
"learning_rate": 7.219999999999999e-07,
|
|
"loss": 3.5762786865234374e-08,
|
|
"num_tokens": 396707.0,
|
|
"reward": -0.08986250013113022,
|
|
"reward_std": 0.10776186282746494,
|
|
"rewards/_reward/mean": -0.08986250013113022,
|
|
"rewards/_reward/std": 0.10776186282746494,
|
|
"step": 140,
|
|
"step_time": 11.505186360995868
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.16601159889250994,
|
|
"epoch": 0.015,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 2.2357609272003174,
|
|
"learning_rate": 7.019999999999999e-07,
|
|
"loss": 1.4007091522216797e-07,
|
|
"num_tokens": 425139.0,
|
|
"reward": -0.06081249937415123,
|
|
"reward_std": 0.1219825193285942,
|
|
"rewards/_reward/mean": -0.06081249937415123,
|
|
"rewards/_reward/std": 0.1219825193285942,
|
|
"step": 150,
|
|
"step_time": 11.64991204040416
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.20173385739326477,
|
|
"epoch": 0.016,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 2.7504308223724365,
|
|
"learning_rate": 6.82e-07,
|
|
"loss": -8.046627044677735e-08,
|
|
"num_tokens": 453235.0,
|
|
"reward": 0.0051499992609024044,
|
|
"reward_std": 0.15129406368359924,
|
|
"rewards/_reward/mean": 0.0051499992609024044,
|
|
"rewards/_reward/std": 0.15129406368359924,
|
|
"step": 160,
|
|
"step_time": 11.575167022497045
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.17478586584329606,
|
|
"epoch": 0.017,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 2.29221773147583,
|
|
"learning_rate": 6.62e-07,
|
|
"loss": 3.725290298461914e-08,
|
|
"num_tokens": 480979.0,
|
|
"reward": -0.042637499421834944,
|
|
"reward_std": 0.07295619142241776,
|
|
"rewards/_reward/mean": -0.042637499421834944,
|
|
"rewards/_reward/std": 0.07295619142241776,
|
|
"step": 170,
|
|
"step_time": 11.508803388095112
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.22443594820797444,
|
|
"epoch": 0.018,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 3.2347171306610107,
|
|
"learning_rate": 6.42e-07,
|
|
"loss": -1.1026859283447266e-07,
|
|
"num_tokens": 508907.0,
|
|
"reward": -0.1161250002682209,
|
|
"reward_std": 0.05800414513796568,
|
|
"rewards/_reward/mean": -0.1161250002682209,
|
|
"rewards/_reward/std": 0.05800414513796568,
|
|
"step": 180,
|
|
"step_time": 11.628456287202425
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.17592381052672862,
|
|
"epoch": 0.019,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 2.917639970779419,
|
|
"learning_rate": 6.219999999999999e-07,
|
|
"loss": -4.395842552185059e-08,
|
|
"num_tokens": 537123.0,
|
|
"reward": -0.06955000124871731,
|
|
"reward_std": 0.12644521035254003,
|
|
"rewards/_reward/mean": -0.06955000124871731,
|
|
"rewards/_reward/std": 0.12644521035254003,
|
|
"step": 190,
|
|
"step_time": 11.57307404029416
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.1488013807684183,
|
|
"epoch": 0.02,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 2.9038145542144775,
|
|
"learning_rate": 6.019999999999999e-07,
|
|
"loss": -1.1101365089416504e-07,
|
|
"num_tokens": 565155.0,
|
|
"reward": -0.12023750096559524,
|
|
"reward_std": 0.010919334553182124,
|
|
"rewards/_reward/mean": -0.12023750096559524,
|
|
"rewards/_reward/std": 0.010919334553182124,
|
|
"step": 200,
|
|
"step_time": 11.57396563780203
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.19087741933763028,
|
|
"epoch": 0.021,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 2.828523874282837,
|
|
"learning_rate": 5.819999999999999e-07,
|
|
"loss": -1.1026859283447266e-07,
|
|
"num_tokens": 593011.0,
|
|
"reward": -0.12507500275969505,
|
|
"reward_std": 0.028615108272060753,
|
|
"rewards/_reward/mean": -0.12507500275969505,
|
|
"rewards/_reward/std": 0.028615108272060753,
|
|
"step": 210,
|
|
"step_time": 12.10818758100213
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.16332522965967655,
|
|
"epoch": 0.022,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 2.2941064834594727,
|
|
"learning_rate": 5.620000000000001e-07,
|
|
"loss": -9.201467037200927e-08,
|
|
"num_tokens": 623443.0,
|
|
"reward": -0.04131250083446503,
|
|
"reward_std": 0.1392579632345587,
|
|
"rewards/_reward/mean": -0.04131250083446503,
|
|
"rewards/_reward/std": 0.1392579632345587,
|
|
"step": 220,
|
|
"step_time": 11.863228278403403
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.16642205603420734,
|
|
"epoch": 0.023,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 2.8652546405792236,
|
|
"learning_rate": 5.420000000000001e-07,
|
|
"loss": -2.9802322387695312e-08,
|
|
"num_tokens": 651835.0,
|
|
"reward": -0.006312502548098564,
|
|
"reward_std": 0.14235199838876725,
|
|
"rewards/_reward/mean": -0.006312502548098564,
|
|
"rewards/_reward/std": 0.14235199838876725,
|
|
"step": 230,
|
|
"step_time": 12.171389962601825
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.21248361319303513,
|
|
"epoch": 0.024,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 3.370008707046509,
|
|
"learning_rate": 5.22e-07,
|
|
"loss": 1.0281801223754883e-07,
|
|
"num_tokens": 685427.0,
|
|
"reward": -0.08841249942779542,
|
|
"reward_std": 0.11107976362109184,
|
|
"rewards/_reward/mean": -0.08841249942779542,
|
|
"rewards/_reward/std": 0.11107976362109184,
|
|
"step": 240,
|
|
"step_time": 11.702623541106004
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.1631117094308138,
|
|
"epoch": 0.025,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 2.2644002437591553,
|
|
"learning_rate": 5.02e-07,
|
|
"loss": 4.1723251342773435e-08,
|
|
"num_tokens": 713915.0,
|
|
"reward": 0.04688749760389328,
|
|
"reward_std": 0.09012880194932223,
|
|
"rewards/_reward/mean": 0.04688749760389328,
|
|
"rewards/_reward/std": 0.09012880194932223,
|
|
"step": 250,
|
|
"step_time": 11.618486247796682
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.875,
|
|
"completions/max_length": 253.0,
|
|
"completions/max_terminated_length": 47.3,
|
|
"completions/mean_length": 246.7125,
|
|
"completions/mean_terminated_length": 39.7375,
|
|
"completions/min_length": 236.5,
|
|
"completions/min_terminated_length": 31.7,
|
|
"entropy": 0.18974447026848792,
|
|
"epoch": 0.026,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 2.280606269836426,
|
|
"learning_rate": 4.82e-07,
|
|
"loss": 0.0100946843624115,
|
|
"num_tokens": 741100.0,
|
|
"reward": -0.07017500177025796,
|
|
"reward_std": 0.1116799698676914,
|
|
"rewards/_reward/mean": -0.07017500177025796,
|
|
"rewards/_reward/std": 0.1116799698676914,
|
|
"step": 260,
|
|
"step_time": 11.640435569002875
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.9875,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 21.3,
|
|
"completions/mean_length": 255.4625,
|
|
"completions/mean_terminated_length": 21.3,
|
|
"completions/min_length": 251.7,
|
|
"completions/min_terminated_length": 21.3,
|
|
"entropy": 0.255718494951725,
|
|
"epoch": 0.027,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 3.1636722087860107,
|
|
"learning_rate": 4.62e-07,
|
|
"loss": 0.00350225567817688,
|
|
"num_tokens": 768897.0,
|
|
"reward": -0.030937498807907103,
|
|
"reward_std": 0.14211445688270033,
|
|
"rewards/_reward/mean": -0.030937498807907103,
|
|
"rewards/_reward/std": 0.14211445688270033,
|
|
"step": 270,
|
|
"step_time": 11.597638047402143
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.19375687539577485,
|
|
"epoch": 0.028,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 3.0973761081695557,
|
|
"learning_rate": 4.4199999999999996e-07,
|
|
"loss": -2.0265579223632812e-07,
|
|
"num_tokens": 796281.0,
|
|
"reward": -0.0963250022381544,
|
|
"reward_std": 0.044649668782949445,
|
|
"rewards/_reward/mean": -0.0963250022381544,
|
|
"rewards/_reward/std": 0.044649668782949445,
|
|
"step": 280,
|
|
"step_time": 11.893776391106076
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.2594174426048994,
|
|
"epoch": 0.029,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 2.0735323429107666,
|
|
"learning_rate": 4.2199999999999994e-07,
|
|
"loss": -5.21540641784668e-08,
|
|
"num_tokens": 822377.0,
|
|
"reward": -0.04639999940991402,
|
|
"reward_std": 0.09656978868879378,
|
|
"rewards/_reward/mean": -0.04639999940991402,
|
|
"rewards/_reward/std": 0.09656978868879378,
|
|
"step": 290,
|
|
"step_time": 11.615450729100848
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.925,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 49.7,
|
|
"completions/mean_length": 255.1,
|
|
"completions/mean_terminated_length": 48.560000610351565,
|
|
"completions/min_length": 251.4,
|
|
"completions/min_terminated_length": 46.6,
|
|
"entropy": 0.1907284392043948,
|
|
"epoch": 0.03,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 3.095362424850464,
|
|
"learning_rate": 4.02e-07,
|
|
"loss": 0.0033017531037330627,
|
|
"num_tokens": 850529.0,
|
|
"reward": -0.1250375010073185,
|
|
"reward_std": 0.08457564520649612,
|
|
"rewards/_reward/mean": -0.1250375010073185,
|
|
"rewards/_reward/std": 0.08457564520649612,
|
|
"step": 300,
|
|
"step_time": 11.74418184790702
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.17942608781158925,
|
|
"epoch": 0.031,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 3.140031337738037,
|
|
"learning_rate": 3.82e-07,
|
|
"loss": -9.462237358093261e-08,
|
|
"num_tokens": 879713.0,
|
|
"reward": -0.09583750143647193,
|
|
"reward_std": 0.062329388014040886,
|
|
"rewards/_reward/mean": -0.09583750143647193,
|
|
"rewards/_reward/std": 0.062329388014040886,
|
|
"step": 310,
|
|
"step_time": 12.055007661701529
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.9875,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 24.5,
|
|
"completions/mean_length": 255.8625,
|
|
"completions/mean_terminated_length": 24.5,
|
|
"completions/min_length": 254.9,
|
|
"completions/min_terminated_length": 24.5,
|
|
"entropy": 0.14358493592590094,
|
|
"epoch": 0.032,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 2.655355215072632,
|
|
"learning_rate": 3.62e-07,
|
|
"loss": 0.0006616078317165375,
|
|
"num_tokens": 908974.0,
|
|
"reward": -0.03269999995827675,
|
|
"reward_std": 0.10602573482319713,
|
|
"rewards/_reward/mean": -0.03269999995827675,
|
|
"rewards/_reward/std": 0.10602573482319713,
|
|
"step": 320,
|
|
"step_time": 11.952868514193687
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.18845350295305252,
|
|
"epoch": 0.033,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 3.4810824394226074,
|
|
"learning_rate": 3.42e-07,
|
|
"loss": 0.0,
|
|
"num_tokens": 937118.0,
|
|
"reward": 0.044287494570016864,
|
|
"reward_std": 0.060938264592550695,
|
|
"rewards/_reward/mean": 0.044287494570016864,
|
|
"rewards/_reward/std": 0.060938264592550695,
|
|
"step": 330,
|
|
"step_time": 11.613528702702023
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.9375,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 22.2,
|
|
"completions/mean_length": 252.4375,
|
|
"completions/mean_terminated_length": 19.9,
|
|
"completions/min_length": 247.9,
|
|
"completions/min_terminated_length": 17.5,
|
|
"entropy": 0.21791861169040203,
|
|
"epoch": 0.034,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 2.5943644046783447,
|
|
"learning_rate": 3.22e-07,
|
|
"loss": 0.009086345881223678,
|
|
"num_tokens": 966385.0,
|
|
"reward": 0.0007125027477741241,
|
|
"reward_std": 0.12920868811197578,
|
|
"rewards/_reward/mean": 0.0007125027477741241,
|
|
"rewards/_reward/std": 0.12920868811197578,
|
|
"step": 340,
|
|
"step_time": 11.72194930910482
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.9875,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 18.9,
|
|
"completions/mean_length": 255.1625,
|
|
"completions/mean_terminated_length": 18.9,
|
|
"completions/min_length": 249.3,
|
|
"completions/min_terminated_length": 18.9,
|
|
"entropy": 0.23234866317361594,
|
|
"epoch": 0.035,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 3.0298092365264893,
|
|
"learning_rate": 3.02e-07,
|
|
"loss": -0.008306542038917541,
|
|
"num_tokens": 994918.0,
|
|
"reward": -0.04598750025033951,
|
|
"reward_std": 0.127252841508016,
|
|
"rewards/_reward/mean": -0.04598750025033951,
|
|
"rewards/_reward/std": 0.127252841508016,
|
|
"step": 350,
|
|
"step_time": 11.696356770189595
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.2507915152236819,
|
|
"epoch": 0.036,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 2.3644959926605225,
|
|
"learning_rate": 2.8199999999999996e-07,
|
|
"loss": 9.238719940185547e-08,
|
|
"num_tokens": 1022654.0,
|
|
"reward": -0.07061250135302544,
|
|
"reward_std": 0.07958197854459285,
|
|
"rewards/_reward/mean": -0.07061250135302544,
|
|
"rewards/_reward/std": 0.07958197854459285,
|
|
"step": 360,
|
|
"step_time": 11.649006188896601
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.9375,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 49.1,
|
|
"completions/mean_length": 254.525,
|
|
"completions/mean_terminated_length": 45.7,
|
|
"completions/min_length": 246.6,
|
|
"completions/min_terminated_length": 41.8,
|
|
"entropy": 0.1780354470014572,
|
|
"epoch": 0.037,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 3.7164721488952637,
|
|
"learning_rate": 2.62e-07,
|
|
"loss": 0.008141186833381654,
|
|
"num_tokens": 1051616.0,
|
|
"reward": 0.004274993389844895,
|
|
"reward_std": 0.13013869293499739,
|
|
"rewards/_reward/mean": 0.004274993389844895,
|
|
"rewards/_reward/std": 0.13013869293499739,
|
|
"step": 370,
|
|
"step_time": 11.659929870004998
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.15224627051502465,
|
|
"epoch": 0.038,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 3.0679633617401123,
|
|
"learning_rate": 2.4199999999999997e-07,
|
|
"loss": 6.444752216339112e-08,
|
|
"num_tokens": 1082152.0,
|
|
"reward": -0.10256250053644181,
|
|
"reward_std": 0.0596468840027228,
|
|
"rewards/_reward/mean": -0.10256250053644181,
|
|
"rewards/_reward/std": 0.0596468840027228,
|
|
"step": 380,
|
|
"step_time": 11.863759143595235
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.17555545847862958,
|
|
"epoch": 0.039,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 1.578389048576355,
|
|
"learning_rate": 2.22e-07,
|
|
"loss": -1.0132789611816406e-07,
|
|
"num_tokens": 1110008.0,
|
|
"reward": -0.12075000219047069,
|
|
"reward_std": 0.010972304828464985,
|
|
"rewards/_reward/mean": -0.12075000219047069,
|
|
"rewards/_reward/std": 0.010972304828464985,
|
|
"step": 390,
|
|
"step_time": 11.711123501494876
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.14941447451710702,
|
|
"epoch": 0.04,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 2.5108535289764404,
|
|
"learning_rate": 2.02e-07,
|
|
"loss": -9.909272193908692e-08,
|
|
"num_tokens": 1140040.0,
|
|
"reward": -0.10333750173449516,
|
|
"reward_std": 0.013116384530439973,
|
|
"rewards/_reward/mean": -0.10333750173449516,
|
|
"rewards/_reward/std": 0.013116384530439973,
|
|
"step": 400,
|
|
"step_time": 11.676777744301944
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.1525548355653882,
|
|
"epoch": 0.041,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 4.128385066986084,
|
|
"learning_rate": 1.82e-07,
|
|
"loss": 1.4454126358032226e-07,
|
|
"num_tokens": 1170448.0,
|
|
"reward": -0.08016249984502792,
|
|
"reward_std": 0.0795291137881577,
|
|
"rewards/_reward/mean": -0.08016249984502792,
|
|
"rewards/_reward/std": 0.0795291137881577,
|
|
"step": 410,
|
|
"step_time": 12.297140166597092
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.22286633253097535,
|
|
"epoch": 0.042,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 3.007291793823242,
|
|
"learning_rate": 1.62e-07,
|
|
"loss": -1.862645149230957e-08,
|
|
"num_tokens": 1199584.0,
|
|
"reward": -0.09757499992847443,
|
|
"reward_std": 0.0740996248088777,
|
|
"rewards/_reward/mean": -0.09757499992847443,
|
|
"rewards/_reward/std": 0.0740996248088777,
|
|
"step": 420,
|
|
"step_time": 11.899142718903022
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.24193776920437812,
|
|
"epoch": 0.043,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 3.724966287612915,
|
|
"learning_rate": 1.4199999999999997e-07,
|
|
"loss": -1.1473894119262696e-07,
|
|
"num_tokens": 1226832.0,
|
|
"reward": -0.1244625024497509,
|
|
"reward_std": 0.01127410950139165,
|
|
"rewards/_reward/mean": -0.1244625024497509,
|
|
"rewards/_reward/std": 0.01127410950139165,
|
|
"step": 430,
|
|
"step_time": 11.78984572509944
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.2233281958848238,
|
|
"epoch": 0.044,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 2.95318865776062,
|
|
"learning_rate": 1.2199999999999998e-07,
|
|
"loss": -1.4901161193847657e-09,
|
|
"num_tokens": 1254944.0,
|
|
"reward": -0.06746249869465828,
|
|
"reward_std": 0.09618873684667051,
|
|
"rewards/_reward/mean": -0.06746249869465828,
|
|
"rewards/_reward/std": 0.09618873684667051,
|
|
"step": 440,
|
|
"step_time": 11.55629740760487
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.19920143522322178,
|
|
"epoch": 0.045,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 3.3122684955596924,
|
|
"learning_rate": 1.0199999999999999e-07,
|
|
"loss": -9.98377799987793e-08,
|
|
"num_tokens": 1283256.0,
|
|
"reward": -0.11557500138878822,
|
|
"reward_std": 0.025760132633149625,
|
|
"rewards/_reward/mean": -0.11557500138878822,
|
|
"rewards/_reward/std": 0.025760132633149625,
|
|
"step": 450,
|
|
"step_time": 11.560638458307949
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.1775781039148569,
|
|
"epoch": 0.046,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 3.5115034580230713,
|
|
"learning_rate": 8.2e-08,
|
|
"loss": -1.3709068298339843e-07,
|
|
"num_tokens": 1310704.0,
|
|
"reward": -0.026362504437565805,
|
|
"reward_std": 0.09747204910963773,
|
|
"rewards/_reward/mean": -0.026362504437565805,
|
|
"rewards/_reward/std": 0.09747204910963773,
|
|
"step": 460,
|
|
"step_time": 11.709659596002894
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.9875,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 24.0,
|
|
"completions/mean_length": 255.8,
|
|
"completions/mean_terminated_length": 24.0,
|
|
"completions/min_length": 254.4,
|
|
"completions/min_terminated_length": 24.0,
|
|
"entropy": 0.18432039245963097,
|
|
"epoch": 0.047,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 4.197425365447998,
|
|
"learning_rate": 6.2e-08,
|
|
"loss": -0.0004772733896970749,
|
|
"num_tokens": 1340320.0,
|
|
"reward": -0.09371250122785568,
|
|
"reward_std": 0.0444566136226058,
|
|
"rewards/_reward/mean": -0.09371250122785568,
|
|
"rewards/_reward/std": 0.0444566136226058,
|
|
"step": 470,
|
|
"step_time": 11.684765285105096
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.9875,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 20.8,
|
|
"completions/mean_length": 255.4,
|
|
"completions/mean_terminated_length": 20.8,
|
|
"completions/min_length": 251.2,
|
|
"completions/min_terminated_length": 20.8,
|
|
"entropy": 0.18885077368468045,
|
|
"epoch": 0.048,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 3.3366804122924805,
|
|
"learning_rate": 4.2e-08,
|
|
"loss": 0.0034372776746749876,
|
|
"num_tokens": 1366728.0,
|
|
"reward": -0.009375004470348359,
|
|
"reward_std": 0.011106315441429615,
|
|
"rewards/_reward/mean": -0.009375004470348359,
|
|
"rewards/_reward/std": 0.011106315441429615,
|
|
"step": 480,
|
|
"step_time": 11.596776261902415
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.9875,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 21.7,
|
|
"completions/mean_length": 255.5125,
|
|
"completions/mean_terminated_length": 21.7,
|
|
"completions/min_length": 252.1,
|
|
"completions/min_terminated_length": 21.7,
|
|
"entropy": 0.15506133139133454,
|
|
"epoch": 0.049,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 2.8624351024627686,
|
|
"learning_rate": 2.2e-08,
|
|
"loss": 0.0021831102669239042,
|
|
"num_tokens": 1397545.0,
|
|
"reward": -0.06493750177323818,
|
|
"reward_std": 0.09045366649515926,
|
|
"rewards/_reward/mean": -0.06493750177323818,
|
|
"rewards/_reward/std": 0.09045366649515926,
|
|
"step": 490,
|
|
"step_time": 11.930105132705648
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 1.0,
|
|
"completions/max_length": 256.0,
|
|
"completions/max_terminated_length": 0.0,
|
|
"completions/mean_length": 256.0,
|
|
"completions/mean_terminated_length": 0.0,
|
|
"completions/min_length": 256.0,
|
|
"completions/min_terminated_length": 0.0,
|
|
"entropy": 0.26948064677417277,
|
|
"epoch": 0.05,
|
|
"frac_reward_zero_std": 0.0,
|
|
"grad_norm": 3.4680824279785156,
|
|
"learning_rate": 2e-09,
|
|
"loss": 8.046627044677735e-08,
|
|
"num_tokens": 1424705.0,
|
|
"reward": -0.1342124991118908,
|
|
"reward_std": 0.048393824510276316,
|
|
"rewards/_reward/mean": -0.1342124991118908,
|
|
"rewards/_reward/std": 0.048393824510276316,
|
|
"step": 500,
|
|
"step_time": 11.641517098102486
|
|
}
|
|
],
|
|
"logging_steps": 10,
|
|
"max_steps": 500,
|
|
"num_input_tokens_seen": 1424705,
|
|
"num_train_epochs": 1,
|
|
"save_steps": 100,
|
|
"stateful_callbacks": {
|
|
"TrainerControl": {
|
|
"args": {
|
|
"should_epoch_stop": false,
|
|
"should_evaluate": false,
|
|
"should_log": false,
|
|
"should_save": true,
|
|
"should_training_stop": true
|
|
},
|
|
"attributes": {}
|
|
}
|
|
},
|
|
"total_flos": 0.0,
|
|
"train_batch_size": 4,
|
|
"trial_name": null,
|
|
"trial_params": null
|
|
}
|