1875 lines
66 KiB
JSON
1875 lines
66 KiB
JSON
{
|
|
"best_metric": null,
|
|
"best_model_checkpoint": null,
|
|
"epoch": 0.9984,
|
|
"eval_steps": 50,
|
|
"global_step": 312,
|
|
"is_hyper_param_search": false,
|
|
"is_local_process_zero": true,
|
|
"is_world_process_zero": true,
|
|
"log_history": [
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.02890625,
|
|
"completions/max_length": 1536.0,
|
|
"completions/max_terminated_length": 1456.6,
|
|
"completions/mean_length": 168.94833984375,
|
|
"completions/mean_terminated_length": 128.25774536132812,
|
|
"completions/min_length": 2.0,
|
|
"completions/min_terminated_length": 2.0,
|
|
"epoch": 0.016,
|
|
"grad_norm": 0.03971971198916435,
|
|
"learning_rate": 3.1249999999999997e-07,
|
|
"loss": 0.0369,
|
|
"num_tokens": 13430383.0,
|
|
"reward": 0.42822265625,
|
|
"reward_std": 0.3223201155662537,
|
|
"rewards/accuracy_reward": 0.19130859375,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.66513671875,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 5
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0181640625,
|
|
"completions/max_length": 1536.0,
|
|
"completions/max_terminated_length": 1478.4,
|
|
"completions/mean_length": 149.1314453125,
|
|
"completions/mean_terminated_length": 123.49475402832032,
|
|
"completions/min_length": 2.0,
|
|
"completions/min_terminated_length": 2.0,
|
|
"epoch": 0.032,
|
|
"grad_norm": 0.01608453504741192,
|
|
"learning_rate": 6.249999999999999e-07,
|
|
"loss": 0.0318,
|
|
"num_tokens": 26914161.0,
|
|
"reward": 0.477685546875,
|
|
"reward_std": 0.29675968885421755,
|
|
"rewards/accuracy_reward": 0.20576171875,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.749609375,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 10
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.009765625,
|
|
"completions/max_length": 1536.0,
|
|
"completions/max_terminated_length": 1360.2,
|
|
"completions/mean_length": 116.2447265625,
|
|
"completions/mean_terminated_length": 102.26145935058594,
|
|
"completions/min_length": 3.2,
|
|
"completions/min_terminated_length": 3.2,
|
|
"epoch": 0.048,
|
|
"grad_norm": 0.024524839594960213,
|
|
"learning_rate": 9.374999999999999e-07,
|
|
"loss": 0.0369,
|
|
"num_tokens": 40009563.0,
|
|
"reward": 0.58642578125,
|
|
"reward_std": 0.20457804799079896,
|
|
"rewards/accuracy_reward": 0.24951171875,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.92333984375,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 15
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.005078125,
|
|
"completions/max_length": 1536.0,
|
|
"completions/max_terminated_length": 1367.2,
|
|
"completions/mean_length": 84.524609375,
|
|
"completions/mean_terminated_length": 77.11885986328124,
|
|
"completions/min_length": 6.8,
|
|
"completions/min_terminated_length": 6.8,
|
|
"epoch": 0.064,
|
|
"grad_norm": 0.00957680307328701,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0187,
|
|
"num_tokens": 52649815.0,
|
|
"reward": 0.65634765625,
|
|
"reward_std": 0.1586296558380127,
|
|
"rewards/accuracy_reward": 0.332421875,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.9802734375,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 20
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.00068359375,
|
|
"completions/max_length": 1287.4,
|
|
"completions/max_terminated_length": 422.8,
|
|
"completions/mean_length": 70.96943359375,
|
|
"completions/mean_terminated_length": 69.96731414794922,
|
|
"completions/min_length": 14.0,
|
|
"completions/min_terminated_length": 14.0,
|
|
"epoch": 0.08,
|
|
"grad_norm": 0.006326671689748764,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0011,
|
|
"num_tokens": 65166014.0,
|
|
"reward": 0.69560546875,
|
|
"reward_std": 0.11688852310180664,
|
|
"rewards/accuracy_reward": 0.39384765625,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.99736328125,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 25
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0001953125,
|
|
"completions/max_length": 843.8,
|
|
"completions/max_terminated_length": 404.0,
|
|
"completions/mean_length": 73.64716796875,
|
|
"completions/mean_terminated_length": 73.3616714477539,
|
|
"completions/min_length": 15.6,
|
|
"completions/min_terminated_length": 15.6,
|
|
"epoch": 0.096,
|
|
"grad_norm": 0.015297948382794857,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0006,
|
|
"num_tokens": 77821089.0,
|
|
"reward": 0.702392578125,
|
|
"reward_std": 0.1077111542224884,
|
|
"rewards/accuracy_reward": 0.4056640625,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.99912109375,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 30
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.000390625,
|
|
"completions/max_length": 1045.2,
|
|
"completions/max_terminated_length": 579.6,
|
|
"completions/mean_length": 77.37705078125,
|
|
"completions/mean_terminated_length": 76.80772857666015,
|
|
"completions/min_length": 15.4,
|
|
"completions/min_terminated_length": 15.4,
|
|
"epoch": 0.112,
|
|
"grad_norm": 0.0034716154914349318,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0027,
|
|
"num_tokens": 90579222.0,
|
|
"reward": 0.717138671875,
|
|
"reward_std": 0.10914339721202851,
|
|
"rewards/accuracy_reward": 0.4361328125,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.99814453125,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 35
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.000390625,
|
|
"completions/max_length": 1137.8,
|
|
"completions/max_terminated_length": 429.2,
|
|
"completions/mean_length": 80.57646484375,
|
|
"completions/mean_terminated_length": 80.00765838623047,
|
|
"completions/min_length": 20.2,
|
|
"completions/min_terminated_length": 20.2,
|
|
"epoch": 0.128,
|
|
"grad_norm": 0.0017264570342376828,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0021,
|
|
"num_tokens": 103177317.0,
|
|
"reward": 0.7154296875,
|
|
"reward_std": 0.09922275692224503,
|
|
"rewards/accuracy_reward": 0.4318359375,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.9990234375,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 40
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.00029296875,
|
|
"completions/max_length": 1027.8,
|
|
"completions/max_terminated_length": 526.4,
|
|
"completions/mean_length": 80.383203125,
|
|
"completions/mean_terminated_length": 79.95665893554687,
|
|
"completions/min_length": 18.2,
|
|
"completions/min_terminated_length": 18.2,
|
|
"epoch": 0.144,
|
|
"grad_norm": 0.0022848467342555523,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0006,
|
|
"num_tokens": 115807193.0,
|
|
"reward": 0.76044921875,
|
|
"reward_std": 0.09742432534694671,
|
|
"rewards/accuracy_reward": 0.5216796875,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.99921875,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 45
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.00048828125,
|
|
"completions/max_length": 1354.8,
|
|
"completions/max_terminated_length": 391.2,
|
|
"completions/mean_length": 85.2048828125,
|
|
"completions/mean_terminated_length": 84.49549560546875,
|
|
"completions/min_length": 29.2,
|
|
"completions/min_terminated_length": 29.2,
|
|
"epoch": 0.16,
|
|
"grad_norm": 0.0022386491764336824,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0009,
|
|
"num_tokens": 128556939.0,
|
|
"reward": 0.7416015625,
|
|
"reward_std": 0.09557278156280517,
|
|
"rewards/accuracy_reward": 0.48388671875,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.99931640625,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 50
|
|
},
|
|
{
|
|
"epoch": 0.16,
|
|
"eval_completions/clipped_ratio": 0.0,
|
|
"eval_completions/max_length": 315.75,
|
|
"eval_completions/max_terminated_length": 315.75,
|
|
"eval_completions/mean_length": 93.1332836151123,
|
|
"eval_completions/mean_terminated_length": 93.1332836151123,
|
|
"eval_completions/min_length": 39.25,
|
|
"eval_completions/min_terminated_length": 39.25,
|
|
"eval_loss": 0.0,
|
|
"eval_num_tokens": 128556939.0,
|
|
"eval_reward": 0.6962890625,
|
|
"eval_reward_std": 0.2430882230401039,
|
|
"eval_rewards/accuracy_reward": 0.392578125,
|
|
"eval_rewards/brier_reward": 0.0,
|
|
"eval_rewards/confidence_one_or_zero": 0.0,
|
|
"eval_rewards/format_reward": 1.0,
|
|
"eval_rewards/mean_confidence_reward": 0.0,
|
|
"eval_runtime": 17.5973,
|
|
"eval_samples_per_second": 28.413,
|
|
"eval_steps_per_second": 0.227,
|
|
"step": 50
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.00068359375,
|
|
"completions/max_length": 1087.0,
|
|
"completions/max_terminated_length": 626.8,
|
|
"completions/mean_length": 93.46279296875,
|
|
"completions/mean_terminated_length": 92.47601928710938,
|
|
"completions/min_length": 30.6,
|
|
"completions/min_terminated_length": 30.6,
|
|
"epoch": 0.176,
|
|
"grad_norm": 0.002047221176326275,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0034,
|
|
"num_tokens": 141607438.0,
|
|
"reward": 0.7376953125,
|
|
"reward_std": 0.09575197547674179,
|
|
"rewards/accuracy_reward": 0.4763671875,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.9990234375,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 55
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.001171875,
|
|
"completions/max_length": 1536.0,
|
|
"completions/max_terminated_length": 727.4,
|
|
"completions/mean_length": 94.7880859375,
|
|
"completions/mean_terminated_length": 93.09821166992188,
|
|
"completions/min_length": 30.8,
|
|
"completions/min_terminated_length": 30.8,
|
|
"epoch": 0.192,
|
|
"grad_norm": 0.0016944622620940208,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0038,
|
|
"num_tokens": 154249204.0,
|
|
"reward": 0.7494140625,
|
|
"reward_std": 0.08774578422307969,
|
|
"rewards/accuracy_reward": 0.5001953125,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.9986328125,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 60
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0001953125,
|
|
"completions/max_length": 945.4,
|
|
"completions/max_terminated_length": 503.0,
|
|
"completions/mean_length": 94.95087890625,
|
|
"completions/mean_terminated_length": 94.66974792480468,
|
|
"completions/min_length": 33.8,
|
|
"completions/min_terminated_length": 33.8,
|
|
"epoch": 0.208,
|
|
"grad_norm": 0.0016015366418287158,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0015,
|
|
"num_tokens": 167110045.0,
|
|
"reward": 0.77373046875,
|
|
"reward_std": 0.08670773208141327,
|
|
"rewards/accuracy_reward": 0.5478515625,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.999609375,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 65
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.001953125,
|
|
"completions/max_length": 1297.4,
|
|
"completions/max_terminated_length": 481.2,
|
|
"completions/mean_length": 96.69267578125,
|
|
"completions/mean_terminated_length": 93.87931976318359,
|
|
"completions/min_length": 32.4,
|
|
"completions/min_terminated_length": 32.4,
|
|
"epoch": 0.224,
|
|
"grad_norm": 0.001440020278096199,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0051,
|
|
"num_tokens": 180109682.0,
|
|
"reward": 0.74384765625,
|
|
"reward_std": 0.08273749649524689,
|
|
"rewards/accuracy_reward": 0.48974609375,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.99794921875,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 70
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.00078125,
|
|
"completions/max_length": 1132.0,
|
|
"completions/max_terminated_length": 595.0,
|
|
"completions/mean_length": 96.2376953125,
|
|
"completions/mean_terminated_length": 95.11122589111328,
|
|
"completions/min_length": 32.6,
|
|
"completions/min_terminated_length": 32.6,
|
|
"epoch": 0.24,
|
|
"grad_norm": 0.0018575695576146245,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0032,
|
|
"num_tokens": 193203156.0,
|
|
"reward": 0.77578125,
|
|
"reward_std": 0.09852926135063171,
|
|
"rewards/accuracy_reward": 0.55234375,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.99921875,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 75
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.000390625,
|
|
"completions/max_length": 1310.2,
|
|
"completions/max_terminated_length": 423.2,
|
|
"completions/mean_length": 91.51962890625,
|
|
"completions/mean_terminated_length": 90.95541381835938,
|
|
"completions/min_length": 33.6,
|
|
"completions/min_terminated_length": 33.6,
|
|
"epoch": 0.256,
|
|
"grad_norm": 0.0016601247480139136,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0016,
|
|
"num_tokens": 206051453.0,
|
|
"reward": 0.76103515625,
|
|
"reward_std": 0.08267472535371781,
|
|
"rewards/accuracy_reward": 0.52265625,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.9994140625,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 80
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.00029296875,
|
|
"completions/max_length": 806.6,
|
|
"completions/max_terminated_length": 347.2,
|
|
"completions/mean_length": 92.82333984375,
|
|
"completions/mean_terminated_length": 92.40117645263672,
|
|
"completions/min_length": 33.0,
|
|
"completions/min_terminated_length": 33.0,
|
|
"epoch": 0.272,
|
|
"grad_norm": 0.0015578230377286673,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0015,
|
|
"num_tokens": 218823980.0,
|
|
"reward": 0.758251953125,
|
|
"reward_std": 0.07986692190170289,
|
|
"rewards/accuracy_reward": 0.516796875,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.99970703125,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 85
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0001953125,
|
|
"completions/max_length": 742.4,
|
|
"completions/max_terminated_length": 510.6,
|
|
"completions/mean_length": 87.11962890625,
|
|
"completions/mean_terminated_length": 86.83695526123047,
|
|
"completions/min_length": 31.4,
|
|
"completions/min_terminated_length": 31.4,
|
|
"epoch": 0.288,
|
|
"grad_norm": 0.0017029134323820472,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.001,
|
|
"num_tokens": 231530581.0,
|
|
"reward": 0.766064453125,
|
|
"reward_std": 0.08002678900957108,
|
|
"rewards/accuracy_reward": 0.53232421875,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.9998046875,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 90
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.000390625,
|
|
"completions/max_length": 1143.6,
|
|
"completions/max_terminated_length": 458.8,
|
|
"completions/mean_length": 87.85400390625,
|
|
"completions/mean_terminated_length": 87.2877700805664,
|
|
"completions/min_length": 32.6,
|
|
"completions/min_terminated_length": 32.6,
|
|
"epoch": 0.304,
|
|
"grad_norm": 0.002200548769906163,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0017,
|
|
"num_tokens": 244216478.0,
|
|
"reward": 0.763037109375,
|
|
"reward_std": 0.07723365724086761,
|
|
"rewards/accuracy_reward": 0.52646484375,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.999609375,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 95
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.00048828125,
|
|
"completions/max_length": 1041.0,
|
|
"completions/max_terminated_length": 307.2,
|
|
"completions/mean_length": 87.27138671875,
|
|
"completions/mean_terminated_length": 86.5638931274414,
|
|
"completions/min_length": 36.2,
|
|
"completions/min_terminated_length": 36.2,
|
|
"epoch": 0.32,
|
|
"grad_norm": 0.0014393636956810951,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0011,
|
|
"num_tokens": 257055161.0,
|
|
"reward": 0.769873046875,
|
|
"reward_std": 0.06611677706241607,
|
|
"rewards/accuracy_reward": 0.54033203125,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.9994140625,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 100
|
|
},
|
|
{
|
|
"epoch": 0.32,
|
|
"eval_completions/clipped_ratio": 0.001953125,
|
|
"eval_completions/max_length": 593.5,
|
|
"eval_completions/max_terminated_length": 258.0,
|
|
"eval_completions/mean_length": 95.18911552429199,
|
|
"eval_completions/mean_terminated_length": 92.36835670471191,
|
|
"eval_completions/min_length": 46.75,
|
|
"eval_completions/min_terminated_length": 46.75,
|
|
"eval_loss": 0.0,
|
|
"eval_num_tokens": 257055161.0,
|
|
"eval_reward": 0.697265625,
|
|
"eval_reward_std": 0.24716106429696083,
|
|
"eval_rewards/accuracy_reward": 0.396484375,
|
|
"eval_rewards/brier_reward": 0.0,
|
|
"eval_rewards/confidence_one_or_zero": 0.0,
|
|
"eval_rewards/format_reward": 0.998046875,
|
|
"eval_rewards/mean_confidence_reward": 0.0,
|
|
"eval_runtime": 24.5152,
|
|
"eval_samples_per_second": 20.395,
|
|
"eval_steps_per_second": 0.163,
|
|
"step": 100
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.00048828125,
|
|
"completions/max_length": 1300.4,
|
|
"completions/max_terminated_length": 441.4,
|
|
"completions/mean_length": 90.16708984375,
|
|
"completions/mean_terminated_length": 89.46054534912109,
|
|
"completions/min_length": 37.2,
|
|
"completions/min_terminated_length": 37.2,
|
|
"epoch": 0.336,
|
|
"grad_norm": 0.0015561841428279877,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0017,
|
|
"num_tokens": 269557224.0,
|
|
"reward": 0.775537109375,
|
|
"reward_std": 0.07957550585269928,
|
|
"rewards/accuracy_reward": 0.5515625,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.99951171875,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 105
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0001953125,
|
|
"completions/max_length": 966.8,
|
|
"completions/max_terminated_length": 503.4,
|
|
"completions/mean_length": 91.57109375,
|
|
"completions/mean_terminated_length": 91.28900604248047,
|
|
"completions/min_length": 37.4,
|
|
"completions/min_terminated_length": 37.4,
|
|
"epoch": 0.352,
|
|
"grad_norm": 0.0015385675942525268,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0015,
|
|
"num_tokens": 282611648.0,
|
|
"reward": 0.749658203125,
|
|
"reward_std": 0.07262765616178513,
|
|
"rewards/accuracy_reward": 0.499609375,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.99970703125,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 110
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0005859375,
|
|
"completions/max_length": 878.6,
|
|
"completions/max_terminated_length": 388.2,
|
|
"completions/mean_length": 92.1431640625,
|
|
"completions/mean_terminated_length": 91.29630737304687,
|
|
"completions/min_length": 36.0,
|
|
"completions/min_terminated_length": 36.0,
|
|
"epoch": 0.368,
|
|
"grad_norm": 0.001171114738099277,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0016,
|
|
"num_tokens": 295476986.0,
|
|
"reward": 0.756591796875,
|
|
"reward_std": 0.06191314309835434,
|
|
"rewards/accuracy_reward": 0.51376953125,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.9994140625,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 115
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0001953125,
|
|
"completions/max_length": 655.4,
|
|
"completions/max_terminated_length": 408.8,
|
|
"completions/mean_length": 93.15146484375,
|
|
"completions/mean_terminated_length": 92.86930847167969,
|
|
"completions/min_length": 36.4,
|
|
"completions/min_terminated_length": 36.4,
|
|
"epoch": 0.384,
|
|
"grad_norm": 0.0033440215047448874,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0004,
|
|
"num_tokens": 308143689.0,
|
|
"reward": 0.772265625,
|
|
"reward_std": 0.06511491686105728,
|
|
"rewards/accuracy_reward": 0.5447265625,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.9998046875,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 120
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0,
|
|
"completions/max_length": 427.8,
|
|
"completions/max_terminated_length": 427.8,
|
|
"completions/mean_length": 91.9720703125,
|
|
"completions/mean_terminated_length": 91.9720703125,
|
|
"completions/min_length": 36.0,
|
|
"completions/min_terminated_length": 36.0,
|
|
"epoch": 0.4,
|
|
"grad_norm": 0.0016430045943707228,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0005,
|
|
"num_tokens": 320978251.0,
|
|
"reward": 0.762353515625,
|
|
"reward_std": 0.0696032926440239,
|
|
"rewards/accuracy_reward": 0.5248046875,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.99990234375,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 125
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.000390625,
|
|
"completions/max_length": 867.8,
|
|
"completions/max_terminated_length": 388.2,
|
|
"completions/mean_length": 94.73369140625,
|
|
"completions/mean_terminated_length": 94.17031402587891,
|
|
"completions/min_length": 40.0,
|
|
"completions/min_terminated_length": 40.0,
|
|
"epoch": 0.416,
|
|
"grad_norm": 0.0013333384413272142,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0015,
|
|
"num_tokens": 333685828.0,
|
|
"reward": 0.769775390625,
|
|
"reward_std": 0.0636753872036934,
|
|
"rewards/accuracy_reward": 0.53994140625,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.999609375,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 130
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.000390625,
|
|
"completions/max_length": 1081.2,
|
|
"completions/max_terminated_length": 420.8,
|
|
"completions/mean_length": 94.04560546875,
|
|
"completions/mean_terminated_length": 93.48209533691406,
|
|
"completions/min_length": 42.6,
|
|
"completions/min_terminated_length": 42.6,
|
|
"epoch": 0.432,
|
|
"grad_norm": 0.0019940051715821028,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0017,
|
|
"num_tokens": 346519511.0,
|
|
"reward": 0.787548828125,
|
|
"reward_std": 0.061328309774398806,
|
|
"rewards/accuracy_reward": 0.57548828125,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.999609375,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 135
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.00029296875,
|
|
"completions/max_length": 1084.4,
|
|
"completions/max_terminated_length": 401.0,
|
|
"completions/mean_length": 95.8486328125,
|
|
"completions/mean_terminated_length": 95.42702941894531,
|
|
"completions/min_length": 41.6,
|
|
"completions/min_terminated_length": 41.6,
|
|
"epoch": 0.448,
|
|
"grad_norm": 0.0024249772541224957,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0015,
|
|
"num_tokens": 359310121.0,
|
|
"reward": 0.766064453125,
|
|
"reward_std": 0.061192930489778516,
|
|
"rewards/accuracy_reward": 0.53251953125,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.999609375,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 140
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.00078125,
|
|
"completions/max_length": 1311.0,
|
|
"completions/max_terminated_length": 497.0,
|
|
"completions/mean_length": 97.95634765625,
|
|
"completions/mean_terminated_length": 96.83337860107422,
|
|
"completions/min_length": 41.4,
|
|
"completions/min_terminated_length": 41.4,
|
|
"epoch": 0.464,
|
|
"grad_norm": 0.0012390408664941788,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0022,
|
|
"num_tokens": 372340330.0,
|
|
"reward": 0.73681640625,
|
|
"reward_std": 0.05486533492803573,
|
|
"rewards/accuracy_reward": 0.47451171875,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.99912109375,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 145
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.000390625,
|
|
"completions/max_length": 1080.2,
|
|
"completions/max_terminated_length": 732.2,
|
|
"completions/mean_length": 98.50205078125,
|
|
"completions/mean_terminated_length": 97.9401870727539,
|
|
"completions/min_length": 41.6,
|
|
"completions/min_terminated_length": 41.6,
|
|
"epoch": 0.48,
|
|
"grad_norm": 0.0014968572650104761,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0015,
|
|
"num_tokens": 385253343.0,
|
|
"reward": 0.77099609375,
|
|
"reward_std": 0.07616502344608307,
|
|
"rewards/accuracy_reward": 0.5423828125,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.999609375,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 150
|
|
},
|
|
{
|
|
"epoch": 0.48,
|
|
"eval_completions/clipped_ratio": 0.0,
|
|
"eval_completions/max_length": 291.5,
|
|
"eval_completions/max_terminated_length": 291.5,
|
|
"eval_completions/mean_length": 97.19706344604492,
|
|
"eval_completions/mean_terminated_length": 97.19706344604492,
|
|
"eval_completions/min_length": 53.75,
|
|
"eval_completions/min_terminated_length": 53.75,
|
|
"eval_loss": 0.0,
|
|
"eval_num_tokens": 385253343.0,
|
|
"eval_reward": 0.724609375,
|
|
"eval_reward_std": 0.24863140657544136,
|
|
"eval_rewards/accuracy_reward": 0.44921875,
|
|
"eval_rewards/brier_reward": 0.0,
|
|
"eval_rewards/confidence_one_or_zero": 0.0,
|
|
"eval_rewards/format_reward": 1.0,
|
|
"eval_rewards/mean_confidence_reward": 0.0,
|
|
"eval_runtime": 15.7439,
|
|
"eval_samples_per_second": 31.758,
|
|
"eval_steps_per_second": 0.254,
|
|
"step": 150
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.00078125,
|
|
"completions/max_length": 1303.2,
|
|
"completions/max_terminated_length": 494.6,
|
|
"completions/mean_length": 98.003125,
|
|
"completions/mean_terminated_length": 96.87923278808594,
|
|
"completions/min_length": 40.4,
|
|
"completions/min_terminated_length": 40.4,
|
|
"epoch": 0.496,
|
|
"grad_norm": 0.0016110693104565144,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0032,
|
|
"num_tokens": 398421055.0,
|
|
"reward": 0.779296875,
|
|
"reward_std": 0.06850271001458168,
|
|
"rewards/accuracy_reward": 0.559375,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.99921875,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 155
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.000390625,
|
|
"completions/max_length": 899.8,
|
|
"completions/max_terminated_length": 518.4,
|
|
"completions/mean_length": 93.11689453125,
|
|
"completions/mean_terminated_length": 92.55267181396485,
|
|
"completions/min_length": 36.0,
|
|
"completions/min_terminated_length": 36.0,
|
|
"epoch": 0.512,
|
|
"grad_norm": 0.0013594066258519888,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0022,
|
|
"num_tokens": 411376556.0,
|
|
"reward": 0.779443359375,
|
|
"reward_std": 0.06428035944700242,
|
|
"rewards/accuracy_reward": 0.55927734375,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.999609375,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 160
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0005859375,
|
|
"completions/max_length": 1116.4,
|
|
"completions/max_terminated_length": 432.0,
|
|
"completions/mean_length": 93.2267578125,
|
|
"completions/mean_terminated_length": 92.3808364868164,
|
|
"completions/min_length": 38.0,
|
|
"completions/min_terminated_length": 38.0,
|
|
"epoch": 0.528,
|
|
"grad_norm": 0.0014900276437401772,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0026,
|
|
"num_tokens": 424217054.0,
|
|
"reward": 0.778076171875,
|
|
"reward_std": 0.0658962957561016,
|
|
"rewards/accuracy_reward": 0.55673828125,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.9994140625,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 165
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0005859375,
|
|
"completions/max_length": 821.4,
|
|
"completions/max_terminated_length": 519.6,
|
|
"completions/mean_length": 93.21083984375,
|
|
"completions/mean_terminated_length": 92.36523132324218,
|
|
"completions/min_length": 35.2,
|
|
"completions/min_terminated_length": 35.2,
|
|
"epoch": 0.544,
|
|
"grad_norm": 0.002942045219242573,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0023,
|
|
"num_tokens": 437191437.0,
|
|
"reward": 0.79521484375,
|
|
"reward_std": 0.07868360131978988,
|
|
"rewards/accuracy_reward": 0.59111328125,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.99931640625,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 170
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0005859375,
|
|
"completions/max_length": 1312.2,
|
|
"completions/max_terminated_length": 406.6,
|
|
"completions/mean_length": 93.5953125,
|
|
"completions/mean_terminated_length": 92.75013885498046,
|
|
"completions/min_length": 37.6,
|
|
"completions/min_terminated_length": 37.6,
|
|
"epoch": 0.56,
|
|
"grad_norm": 0.0016339273424819112,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0019,
|
|
"num_tokens": 449827581.0,
|
|
"reward": 0.768896484375,
|
|
"reward_std": 0.06067224889993668,
|
|
"rewards/accuracy_reward": 0.53837890625,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.9994140625,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 175
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.00107421875,
|
|
"completions/max_length": 1292.6,
|
|
"completions/max_terminated_length": 394.6,
|
|
"completions/mean_length": 94.48125,
|
|
"completions/mean_terminated_length": 92.93127746582032,
|
|
"completions/min_length": 43.8,
|
|
"completions/min_terminated_length": 43.8,
|
|
"epoch": 0.576,
|
|
"grad_norm": 0.0014951933408156037,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0032,
|
|
"num_tokens": 462838013.0,
|
|
"reward": 0.767529296875,
|
|
"reward_std": 0.05619642436504364,
|
|
"rewards/accuracy_reward": 0.5361328125,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.99892578125,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 180
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0001953125,
|
|
"completions/max_length": 590.2,
|
|
"completions/max_terminated_length": 341.8,
|
|
"completions/mean_length": 90.8630859375,
|
|
"completions/mean_terminated_length": 90.5811050415039,
|
|
"completions/min_length": 38.6,
|
|
"completions/min_terminated_length": 38.6,
|
|
"epoch": 0.592,
|
|
"grad_norm": 0.0016095780301839113,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0011,
|
|
"num_tokens": 475792483.0,
|
|
"reward": 0.7609375,
|
|
"reward_std": 0.05783471763134003,
|
|
"rewards/accuracy_reward": 0.5220703125,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.9998046875,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 185
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0005859375,
|
|
"completions/max_length": 1291.2,
|
|
"completions/max_terminated_length": 429.2,
|
|
"completions/mean_length": 92.3296875,
|
|
"completions/mean_terminated_length": 91.4835708618164,
|
|
"completions/min_length": 40.4,
|
|
"completions/min_terminated_length": 40.4,
|
|
"epoch": 0.608,
|
|
"grad_norm": 0.0014157961122691631,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0017,
|
|
"num_tokens": 488593747.0,
|
|
"reward": 0.775439453125,
|
|
"reward_std": 0.0586208701133728,
|
|
"rewards/accuracy_reward": 0.55146484375,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.9994140625,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 190
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.00029296875,
|
|
"completions/max_length": 938.8,
|
|
"completions/max_terminated_length": 530.8,
|
|
"completions/mean_length": 91.69248046875,
|
|
"completions/mean_terminated_length": 91.26978149414063,
|
|
"completions/min_length": 40.0,
|
|
"completions/min_terminated_length": 40.0,
|
|
"epoch": 0.624,
|
|
"grad_norm": 0.0018766943830996752,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0012,
|
|
"num_tokens": 501732902.0,
|
|
"reward": 0.776904296875,
|
|
"reward_std": 0.05459159761667252,
|
|
"rewards/accuracy_reward": 0.5541015625,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.99970703125,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 195
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.00048828125,
|
|
"completions/max_length": 1351.0,
|
|
"completions/max_terminated_length": 483.4,
|
|
"completions/mean_length": 92.82841796875,
|
|
"completions/mean_terminated_length": 92.12330017089843,
|
|
"completions/min_length": 41.0,
|
|
"completions/min_terminated_length": 41.0,
|
|
"epoch": 0.64,
|
|
"grad_norm": 0.0021018637344241142,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0016,
|
|
"num_tokens": 514882473.0,
|
|
"reward": 0.793505859375,
|
|
"reward_std": 0.05022450163960457,
|
|
"rewards/accuracy_reward": 0.58759765625,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.9994140625,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 200
|
|
},
|
|
{
|
|
"epoch": 0.64,
|
|
"eval_completions/clipped_ratio": 0.0,
|
|
"eval_completions/max_length": 214.75,
|
|
"eval_completions/max_terminated_length": 214.75,
|
|
"eval_completions/mean_length": 91.26717376708984,
|
|
"eval_completions/mean_terminated_length": 91.26717376708984,
|
|
"eval_completions/min_length": 46.0,
|
|
"eval_completions/min_terminated_length": 46.0,
|
|
"eval_loss": 0.0,
|
|
"eval_num_tokens": 514882473.0,
|
|
"eval_reward": 0.7216796875,
|
|
"eval_reward_std": 0.243845384567976,
|
|
"eval_rewards/accuracy_reward": 0.443359375,
|
|
"eval_rewards/brier_reward": 0.0,
|
|
"eval_rewards/confidence_one_or_zero": 0.0,
|
|
"eval_rewards/format_reward": 1.0,
|
|
"eval_rewards/mean_confidence_reward": 0.0,
|
|
"eval_runtime": 13.4425,
|
|
"eval_samples_per_second": 37.195,
|
|
"eval_steps_per_second": 0.298,
|
|
"step": 200
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0001953125,
|
|
"completions/max_length": 848.6,
|
|
"completions/max_terminated_length": 391.0,
|
|
"completions/mean_length": 91.15234375,
|
|
"completions/mean_terminated_length": 90.87013397216796,
|
|
"completions/min_length": 36.6,
|
|
"completions/min_terminated_length": 36.6,
|
|
"epoch": 0.656,
|
|
"grad_norm": 0.002083825645968318,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0004,
|
|
"num_tokens": 527528737.0,
|
|
"reward": 0.761474609375,
|
|
"reward_std": 0.05676820129156113,
|
|
"rewards/accuracy_reward": 0.52314453125,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.9998046875,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 205
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.000390625,
|
|
"completions/max_length": 1066.2,
|
|
"completions/max_terminated_length": 362.2,
|
|
"completions/mean_length": 93.4865234375,
|
|
"completions/mean_terminated_length": 92.92289428710937,
|
|
"completions/min_length": 39.0,
|
|
"completions/min_terminated_length": 39.0,
|
|
"epoch": 0.672,
|
|
"grad_norm": 0.0021750768646597862,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0016,
|
|
"num_tokens": 540255799.0,
|
|
"reward": 0.76650390625,
|
|
"reward_std": 0.05727210938930512,
|
|
"rewards/accuracy_reward": 0.5333984375,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.999609375,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 210
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.00029296875,
|
|
"completions/max_length": 847.6,
|
|
"completions/max_terminated_length": 399.6,
|
|
"completions/mean_length": 96.2537109375,
|
|
"completions/mean_terminated_length": 95.83192443847656,
|
|
"completions/min_length": 39.4,
|
|
"completions/min_terminated_length": 39.4,
|
|
"epoch": 0.688,
|
|
"grad_norm": 0.003001829609274864,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0017,
|
|
"num_tokens": 553051677.0,
|
|
"reward": 0.781494140625,
|
|
"reward_std": 0.05510343983769417,
|
|
"rewards/accuracy_reward": 0.56328125,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.99970703125,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 215
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 9.765625e-05,
|
|
"completions/max_length": 596.2,
|
|
"completions/max_terminated_length": 408.2,
|
|
"completions/mean_length": 93.3064453125,
|
|
"completions/mean_terminated_length": 93.165869140625,
|
|
"completions/min_length": 42.2,
|
|
"completions/min_terminated_length": 42.2,
|
|
"epoch": 0.704,
|
|
"grad_norm": 0.0012618749169632792,
|
|
"learning_rate": 1e-06,
|
|
"loss": -0.0001,
|
|
"num_tokens": 565729599.0,
|
|
"reward": 0.782421875,
|
|
"reward_std": 0.046160271018743516,
|
|
"rewards/accuracy_reward": 0.56494140625,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.99990234375,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 220
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0005859375,
|
|
"completions/max_length": 1132.4,
|
|
"completions/max_terminated_length": 580.2,
|
|
"completions/mean_length": 96.8970703125,
|
|
"completions/mean_terminated_length": 96.05291748046875,
|
|
"completions/min_length": 44.0,
|
|
"completions/min_terminated_length": 44.0,
|
|
"epoch": 0.72,
|
|
"grad_norm": 0.0017941860714927316,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0011,
|
|
"num_tokens": 578588001.0,
|
|
"reward": 0.7890625,
|
|
"reward_std": 0.054417699575424194,
|
|
"rewards/accuracy_reward": 0.5787109375,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.9994140625,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 225
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0005859375,
|
|
"completions/max_length": 832.4,
|
|
"completions/max_terminated_length": 407.2,
|
|
"completions/mean_length": 97.96123046875,
|
|
"completions/mean_terminated_length": 97.11712493896485,
|
|
"completions/min_length": 46.6,
|
|
"completions/min_terminated_length": 46.6,
|
|
"epoch": 0.736,
|
|
"grad_norm": 0.0014221714809536934,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0012,
|
|
"num_tokens": 591387028.0,
|
|
"reward": 0.78837890625,
|
|
"reward_std": 0.048315313458442685,
|
|
"rewards/accuracy_reward": 0.57734375,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.9994140625,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 230
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0001953125,
|
|
"completions/max_length": 631.4,
|
|
"completions/max_terminated_length": 440.4,
|
|
"completions/mean_length": 99.93115234375,
|
|
"completions/mean_terminated_length": 99.65132141113281,
|
|
"completions/min_length": 48.2,
|
|
"completions/min_terminated_length": 48.2,
|
|
"epoch": 0.752,
|
|
"grad_norm": 0.0011550748022273183,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0011,
|
|
"num_tokens": 604493843.0,
|
|
"reward": 0.785107421875,
|
|
"reward_std": 0.052481997013092044,
|
|
"rewards/accuracy_reward": 0.57041015625,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.9998046875,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 235
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.001171875,
|
|
"completions/max_length": 848.6,
|
|
"completions/max_terminated_length": 599.0,
|
|
"completions/mean_length": 105.28837890625,
|
|
"completions/mean_terminated_length": 103.6119171142578,
|
|
"completions/min_length": 46.2,
|
|
"completions/min_terminated_length": 46.2,
|
|
"epoch": 0.768,
|
|
"grad_norm": 0.0019090637797489762,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0023,
|
|
"num_tokens": 617361020.0,
|
|
"reward": 0.76201171875,
|
|
"reward_std": 0.05515236109495163,
|
|
"rewards/accuracy_reward": 0.5251953125,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.998828125,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 240
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.00029296875,
|
|
"completions/max_length": 1095.8,
|
|
"completions/max_terminated_length": 401.2,
|
|
"completions/mean_length": 106.9001953125,
|
|
"completions/mean_terminated_length": 106.48137512207032,
|
|
"completions/min_length": 49.8,
|
|
"completions/min_terminated_length": 49.8,
|
|
"epoch": 0.784,
|
|
"grad_norm": 0.0015024510212242603,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0014,
|
|
"num_tokens": 630486366.0,
|
|
"reward": 0.787255859375,
|
|
"reward_std": 0.05614206939935684,
|
|
"rewards/accuracy_reward": 0.5748046875,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.99970703125,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 245
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.00048828125,
|
|
"completions/max_length": 1434.8,
|
|
"completions/max_terminated_length": 533.6,
|
|
"completions/mean_length": 113.03037109375,
|
|
"completions/mean_terminated_length": 112.33526763916015,
|
|
"completions/min_length": 51.4,
|
|
"completions/min_terminated_length": 51.4,
|
|
"epoch": 0.8,
|
|
"grad_norm": 0.0013259610859677196,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0012,
|
|
"num_tokens": 643510677.0,
|
|
"reward": 0.802490234375,
|
|
"reward_std": 0.04771163538098335,
|
|
"rewards/accuracy_reward": 0.60546875,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.99951171875,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 250
|
|
},
|
|
{
|
|
"epoch": 0.8,
|
|
"eval_completions/clipped_ratio": 0.0,
|
|
"eval_completions/max_length": 298.75,
|
|
"eval_completions/max_terminated_length": 298.75,
|
|
"eval_completions/mean_length": 116.78151893615723,
|
|
"eval_completions/mean_terminated_length": 116.78151893615723,
|
|
"eval_completions/min_length": 58.75,
|
|
"eval_completions/min_terminated_length": 58.75,
|
|
"eval_loss": 0.0,
|
|
"eval_num_tokens": 643510677.0,
|
|
"eval_reward": 0.720703125,
|
|
"eval_reward_std": 0.247483242303133,
|
|
"eval_rewards/accuracy_reward": 0.44140625,
|
|
"eval_rewards/brier_reward": 0.0,
|
|
"eval_rewards/confidence_one_or_zero": 0.0,
|
|
"eval_rewards/format_reward": 1.0,
|
|
"eval_rewards/mean_confidence_reward": 0.0,
|
|
"eval_runtime": 16.3536,
|
|
"eval_samples_per_second": 30.574,
|
|
"eval_steps_per_second": 0.245,
|
|
"step": 250
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.001171875,
|
|
"completions/max_length": 1216.8,
|
|
"completions/max_terminated_length": 613.8,
|
|
"completions/mean_length": 114.61943359375,
|
|
"completions/mean_terminated_length": 112.94878845214843,
|
|
"completions/min_length": 55.2,
|
|
"completions/min_terminated_length": 55.2,
|
|
"epoch": 0.816,
|
|
"grad_norm": 0.0017014788463711739,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0034,
|
|
"num_tokens": 656639868.0,
|
|
"reward": 0.79716796875,
|
|
"reward_std": 0.059028515964746474,
|
|
"rewards/accuracy_reward": 0.59560546875,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.99873046875,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 255
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.00029296875,
|
|
"completions/max_length": 1157.6,
|
|
"completions/max_terminated_length": 482.0,
|
|
"completions/mean_length": 113.28662109375,
|
|
"completions/mean_terminated_length": 112.86976013183593,
|
|
"completions/min_length": 50.2,
|
|
"completions/min_terminated_length": 50.2,
|
|
"epoch": 0.832,
|
|
"grad_norm": 0.0014079577522352338,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0012,
|
|
"num_tokens": 669664595.0,
|
|
"reward": 0.78955078125,
|
|
"reward_std": 0.0593775637447834,
|
|
"rewards/accuracy_reward": 0.57939453125,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.99970703125,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 260
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.00048828125,
|
|
"completions/max_length": 1129.8,
|
|
"completions/max_terminated_length": 445.4,
|
|
"completions/mean_length": 105.9974609375,
|
|
"completions/mean_terminated_length": 105.29950103759765,
|
|
"completions/min_length": 51.2,
|
|
"completions/min_terminated_length": 51.2,
|
|
"epoch": 0.848,
|
|
"grad_norm": 0.002518897643312812,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0019,
|
|
"num_tokens": 682620697.0,
|
|
"reward": 0.77509765625,
|
|
"reward_std": 0.05453771948814392,
|
|
"rewards/accuracy_reward": 0.55068359375,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.99951171875,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 265
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.00029296875,
|
|
"completions/max_length": 841.4,
|
|
"completions/max_terminated_length": 388.6,
|
|
"completions/mean_length": 102.1765625,
|
|
"completions/mean_terminated_length": 101.75653228759765,
|
|
"completions/min_length": 49.8,
|
|
"completions/min_terminated_length": 49.8,
|
|
"epoch": 0.864,
|
|
"grad_norm": 0.002422819845378399,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0015,
|
|
"num_tokens": 695510121.0,
|
|
"reward": 0.808837890625,
|
|
"reward_std": 0.05869279354810715,
|
|
"rewards/accuracy_reward": 0.61796875,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.99970703125,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 270
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.00029296875,
|
|
"completions/max_length": 878.2,
|
|
"completions/max_terminated_length": 402.8,
|
|
"completions/mean_length": 95.89345703125,
|
|
"completions/mean_terminated_length": 95.47228546142578,
|
|
"completions/min_length": 45.4,
|
|
"completions/min_terminated_length": 45.4,
|
|
"epoch": 0.88,
|
|
"grad_norm": 0.002903913613408804,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0017,
|
|
"num_tokens": 708495462.0,
|
|
"reward": 0.7587890625,
|
|
"reward_std": 0.05212245061993599,
|
|
"rewards/accuracy_reward": 0.51796875,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.999609375,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 275
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.00048828125,
|
|
"completions/max_length": 1318.2,
|
|
"completions/max_terminated_length": 486.4,
|
|
"completions/mean_length": 93.419921875,
|
|
"completions/mean_terminated_length": 92.71511383056641,
|
|
"completions/min_length": 43.0,
|
|
"completions/min_terminated_length": 43.0,
|
|
"epoch": 0.896,
|
|
"grad_norm": 0.001861903234384954,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0021,
|
|
"num_tokens": 721419250.0,
|
|
"reward": 0.777197265625,
|
|
"reward_std": 0.046849222481250764,
|
|
"rewards/accuracy_reward": 0.5548828125,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.99951171875,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 280
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.00068359375,
|
|
"completions/max_length": 1306.6,
|
|
"completions/max_terminated_length": 360.6,
|
|
"completions/mean_length": 90.81923828125,
|
|
"completions/mean_terminated_length": 89.83069915771485,
|
|
"completions/min_length": 43.2,
|
|
"completions/min_terminated_length": 43.2,
|
|
"epoch": 0.912,
|
|
"grad_norm": 0.0009662771481089294,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0016,
|
|
"num_tokens": 734256855.0,
|
|
"reward": 0.782080078125,
|
|
"reward_std": 0.04545313939452171,
|
|
"rewards/accuracy_reward": 0.56484375,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.99931640625,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 285
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.00029296875,
|
|
"completions/max_length": 816.4,
|
|
"completions/max_terminated_length": 363.8,
|
|
"completions/mean_length": 86.1810546875,
|
|
"completions/mean_terminated_length": 85.75641479492188,
|
|
"completions/min_length": 37.0,
|
|
"completions/min_terminated_length": 37.0,
|
|
"epoch": 0.928,
|
|
"grad_norm": 0.0012876720866188407,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0012,
|
|
"num_tokens": 747022485.0,
|
|
"reward": 0.772705078125,
|
|
"reward_std": 0.045815450698137285,
|
|
"rewards/accuracy_reward": 0.545703125,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.99970703125,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 290
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0001953125,
|
|
"completions/max_length": 822.2,
|
|
"completions/max_terminated_length": 324.0,
|
|
"completions/mean_length": 83.45751953125,
|
|
"completions/mean_terminated_length": 83.17401275634765,
|
|
"completions/min_length": 40.6,
|
|
"completions/min_terminated_length": 40.6,
|
|
"epoch": 0.944,
|
|
"grad_norm": 0.002664135303348303,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0011,
|
|
"num_tokens": 759708834.0,
|
|
"reward": 0.779052734375,
|
|
"reward_std": 0.05606053844094276,
|
|
"rewards/accuracy_reward": 0.55830078125,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.9998046875,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 295
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.000390625,
|
|
"completions/max_length": 1058.4,
|
|
"completions/max_terminated_length": 550.4,
|
|
"completions/mean_length": 83.72802734375,
|
|
"completions/mean_terminated_length": 83.16080169677734,
|
|
"completions/min_length": 41.0,
|
|
"completions/min_terminated_length": 41.0,
|
|
"epoch": 0.96,
|
|
"grad_norm": 0.0016519302735105157,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0008,
|
|
"num_tokens": 772362849.0,
|
|
"reward": 0.7677734375,
|
|
"reward_std": 0.04194698035717011,
|
|
"rewards/accuracy_reward": 0.5359375,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.999609375,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 300
|
|
},
|
|
{
|
|
"epoch": 0.96,
|
|
"eval_completions/clipped_ratio": 0.0,
|
|
"eval_completions/max_length": 254.0,
|
|
"eval_completions/max_terminated_length": 254.0,
|
|
"eval_completions/mean_length": 84.06842613220215,
|
|
"eval_completions/mean_terminated_length": 84.06842613220215,
|
|
"eval_completions/min_length": 45.25,
|
|
"eval_completions/min_terminated_length": 45.25,
|
|
"eval_loss": 0.0,
|
|
"eval_num_tokens": 772362849.0,
|
|
"eval_reward": 0.72265625,
|
|
"eval_reward_std": 0.2485281154513359,
|
|
"eval_rewards/accuracy_reward": 0.4453125,
|
|
"eval_rewards/brier_reward": 0.0,
|
|
"eval_rewards/confidence_one_or_zero": 0.0,
|
|
"eval_rewards/format_reward": 1.0,
|
|
"eval_rewards/mean_confidence_reward": 0.0,
|
|
"eval_runtime": 14.5394,
|
|
"eval_samples_per_second": 34.389,
|
|
"eval_steps_per_second": 0.275,
|
|
"step": 300
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.000390625,
|
|
"completions/max_length": 1089.0,
|
|
"completions/max_terminated_length": 383.0,
|
|
"completions/mean_length": 83.73583984375,
|
|
"completions/mean_terminated_length": 83.16815948486328,
|
|
"completions/min_length": 41.6,
|
|
"completions/min_terminated_length": 41.6,
|
|
"epoch": 0.976,
|
|
"grad_norm": 0.001557100797072053,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0009,
|
|
"num_tokens": 784937744.0,
|
|
"reward": 0.784521484375,
|
|
"reward_std": 0.051283557713031766,
|
|
"rewards/accuracy_reward": 0.56943359375,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.999609375,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 305
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.00029296875,
|
|
"completions/max_length": 807.2,
|
|
"completions/max_terminated_length": 342.4,
|
|
"completions/mean_length": 85.384375,
|
|
"completions/mean_terminated_length": 84.96044158935547,
|
|
"completions/min_length": 40.6,
|
|
"completions/min_terminated_length": 40.6,
|
|
"epoch": 0.992,
|
|
"grad_norm": 0.0014634578255936503,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.0006,
|
|
"num_tokens": 797796880.0,
|
|
"reward": 0.776220703125,
|
|
"reward_std": 0.04600930213928223,
|
|
"rewards/accuracy_reward": 0.552734375,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.99970703125,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 310
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.000244140625,
|
|
"completions/max_length": 943.0,
|
|
"completions/max_terminated_length": 318.5,
|
|
"completions/mean_length": 85.54854202270508,
|
|
"completions/mean_terminated_length": 85.19485473632812,
|
|
"completions/min_length": 43.0,
|
|
"completions/min_terminated_length": 43.0,
|
|
"epoch": 0.9984,
|
|
"num_tokens": 802903205.0,
|
|
"reward": 0.7908935546875,
|
|
"reward_std": 0.050602177157998085,
|
|
"rewards/accuracy_reward": 0.58251953125,
|
|
"rewards/brier_reward": 0.0,
|
|
"rewards/confidence_one_or_zero": 0.0,
|
|
"rewards/format_reward": 0.999267578125,
|
|
"rewards/mean_confidence_reward": 0.0,
|
|
"step": 312,
|
|
"total_flos": 0.0,
|
|
"train_loss": 0.003547575732227415,
|
|
"train_runtime": 65955.8089,
|
|
"train_samples_per_second": 0.303,
|
|
"train_steps_per_second": 0.005
|
|
}
|
|
],
|
|
"logging_steps": 5,
|
|
"max_steps": 312,
|
|
"num_input_tokens_seen": 802903205,
|
|
"num_train_epochs": 1,
|
|
"save_steps": 60,
|
|
"stateful_callbacks": {
|
|
"TrainerControl": {
|
|
"args": {
|
|
"should_epoch_stop": false,
|
|
"should_evaluate": false,
|
|
"should_log": false,
|
|
"should_save": true,
|
|
"should_training_stop": true
|
|
},
|
|
"attributes": {}
|
|
}
|
|
},
|
|
"total_flos": 0.0,
|
|
"train_batch_size": 8,
|
|
"trial_name": null,
|
|
"trial_params": null
|
|
}
|