1565 lines
65 KiB
JSON
1565 lines
65 KiB
JSON
{
|
|
"best_global_step": 180,
|
|
"best_metric": 57.75,
|
|
"best_model_checkpoint": "Qwen3-0.6B-GRPO-GSM8K-Think-A800-Stable/checkpoint-180",
|
|
"epoch": 0.43938161106590723,
|
|
"eval_steps": 10,
|
|
"global_step": 180,
|
|
"is_hyper_param_search": false,
|
|
"is_local_process_zero": true,
|
|
"is_world_process_zero": true,
|
|
"log_history": [
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.4,
|
|
"completions/max_length": 512.0,
|
|
"completions/max_terminated_length": 500.6,
|
|
"completions/mean_length": 411.5416748046875,
|
|
"completions/mean_terminated_length": 344.4086486816406,
|
|
"completions/min_length": 213.6,
|
|
"completions/min_terminated_length": 213.6,
|
|
"epoch": 0.012205044751830757,
|
|
"frac_reward_zero_std": 0.544444453716278,
|
|
"grad_norm": 0.9040567874908447,
|
|
"kl": 0.0003161211425322108,
|
|
"learning_rate": 9.75609756097561e-08,
|
|
"loss": 0.0242,
|
|
"num_tokens": 262959.0,
|
|
"reward": 3.172777843475342,
|
|
"reward_std": 0.5250407099723816,
|
|
"rewards/correctness_reward_func/mean": 1.75,
|
|
"rewards/correctness_reward_func/std": 1.4723714113235473,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.4227777779102325,
|
|
"rewards/think_structure_reward_func/std": 0.10014611333608628,
|
|
"step": 5
|
|
},
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.3805555555555556,
|
|
"completions/max_length": 512.0,
|
|
"completions/max_terminated_length": 502.6,
|
|
"completions/mean_length": 421.08611450195315,
|
|
"completions/mean_terminated_length": 364.81788330078126,
|
|
"completions/min_length": 229.0,
|
|
"completions/min_terminated_length": 229.0,
|
|
"epoch": 0.024410089503661515,
|
|
"frac_reward_zero_std": 0.42222222685813904,
|
|
"grad_norm": 1.129333734512329,
|
|
"kl": 0.0003199623771555101,
|
|
"learning_rate": 2.195121951219512e-07,
|
|
"loss": 0.0286,
|
|
"num_tokens": 528002.0,
|
|
"reward": 3.1483333110809326,
|
|
"reward_std": 0.6357886552810669,
|
|
"rewards/correctness_reward_func/mean": 1.725,
|
|
"rewards/correctness_reward_func/std": 1.4699805498123169,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.4233333349227905,
|
|
"rewards/think_structure_reward_func/std": 0.1003103420138359,
|
|
"step": 10
|
|
},
|
|
{
|
|
"epoch": 0.024410089503661515,
|
|
"eval_clip_ratio": 0.0,
|
|
"eval_completions/clipped_ratio": 0.4,
|
|
"eval_completions/max_length": 512.0,
|
|
"eval_completions/max_terminated_length": 465.84,
|
|
"eval_completions/mean_length": 420.2975,
|
|
"eval_completions/mean_terminated_length": 361.74707275390625,
|
|
"eval_completions/min_length": 264.24,
|
|
"eval_completions/min_terminated_length": 264.24,
|
|
"eval_frac_reward_zero_std": 0.44,
|
|
"eval_kl": 0.0003276056196773425,
|
|
"eval_loss": 0.028326217085123062,
|
|
"eval_num_tokens": 528002.0,
|
|
"eval_reward": 2.9445000076293946,
|
|
"eval_reward_std": 0.6985072362422943,
|
|
"eval_rewards/correctness_reward_func/mean": 1.5225,
|
|
"eval_rewards/correctness_reward_func/std": 1.4233518791198732,
|
|
"eval_rewards/format_reward_func/mean": 0.5,
|
|
"eval_rewards/format_reward_func/std": 0.0,
|
|
"eval_rewards/int_reward_func/mean": 0.5,
|
|
"eval_rewards/int_reward_func/std": 0.0,
|
|
"eval_rewards/think_structure_reward_func/mean": 0.42199999928474424,
|
|
"eval_rewards/think_structure_reward_func/std": 0.09455148339271545,
|
|
"eval_runtime": 417.3349,
|
|
"eval_samples_per_second": 0.24,
|
|
"eval_steps_per_second": 0.017,
|
|
"step": 10
|
|
},
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.48611111111111116,
|
|
"completions/max_length": 512.0,
|
|
"completions/max_terminated_length": 502.2,
|
|
"completions/mean_length": 433.49444580078125,
|
|
"completions/mean_terminated_length": 358.0120422363281,
|
|
"completions/min_length": 216.0,
|
|
"completions/min_terminated_length": 216.0,
|
|
"epoch": 0.03661513425549227,
|
|
"frac_reward_zero_std": 0.4777777850627899,
|
|
"grad_norm": 1.1167465448379517,
|
|
"kl": 0.00036110289705296357,
|
|
"learning_rate": 3.4146341463414634e-07,
|
|
"loss": 0.0257,
|
|
"num_tokens": 799664.0,
|
|
"reward": 2.853333282470703,
|
|
"reward_std": 0.6509143769741058,
|
|
"rewards/correctness_reward_func/mean": 1.45,
|
|
"rewards/correctness_reward_func/std": 1.4647655248641969,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.4033333480358124,
|
|
"rewards/think_structure_reward_func/std": 0.10041487812995911,
|
|
"step": 15
|
|
},
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.39722222222222225,
|
|
"completions/max_length": 512.0,
|
|
"completions/max_terminated_length": 499.0,
|
|
"completions/mean_length": 416.0944458007813,
|
|
"completions/mean_terminated_length": 353.4386840820313,
|
|
"completions/min_length": 204.0,
|
|
"completions/min_terminated_length": 204.0,
|
|
"epoch": 0.04882017900732303,
|
|
"frac_reward_zero_std": 0.5000000119209289,
|
|
"grad_norm": 1.0495957136154175,
|
|
"kl": 0.0006103336068917997,
|
|
"learning_rate": 4.634146341463415e-07,
|
|
"loss": 0.0203,
|
|
"num_tokens": 1062382.0,
|
|
"reward": 3.046666717529297,
|
|
"reward_std": 0.6736284494400024,
|
|
"rewards/correctness_reward_func/mean": 1.625,
|
|
"rewards/correctness_reward_func/std": 1.4933619976043702,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.42166666984558104,
|
|
"rewards/think_structure_reward_func/std": 0.0961018055677414,
|
|
"step": 20
|
|
},
|
|
{
|
|
"epoch": 0.04882017900732303,
|
|
"eval_clip_ratio": 0.0,
|
|
"eval_completions/clipped_ratio": 0.4,
|
|
"eval_completions/max_length": 512.0,
|
|
"eval_completions/max_terminated_length": 472.0,
|
|
"eval_completions/mean_length": 419.9625,
|
|
"eval_completions/mean_terminated_length": 359.3039898681641,
|
|
"eval_completions/min_length": 255.84,
|
|
"eval_completions/min_terminated_length": 255.84,
|
|
"eval_frac_reward_zero_std": 0.46,
|
|
"eval_kl": 0.0010042595444247127,
|
|
"eval_loss": 0.03167510777711868,
|
|
"eval_num_tokens": 1062382.0,
|
|
"eval_reward": 3.010999994277954,
|
|
"eval_reward_std": 0.5388501909375191,
|
|
"eval_rewards/correctness_reward_func/mean": 1.59,
|
|
"eval_rewards/correctness_reward_func/std": 1.3927859783172607,
|
|
"eval_rewards/format_reward_func/mean": 0.5,
|
|
"eval_rewards/format_reward_func/std": 0.0,
|
|
"eval_rewards/int_reward_func/mean": 0.5,
|
|
"eval_rewards/int_reward_func/std": 0.0,
|
|
"eval_rewards/think_structure_reward_func/mean": 0.4210000026226044,
|
|
"eval_rewards/think_structure_reward_func/std": 0.09440703004598618,
|
|
"eval_runtime": 413.9411,
|
|
"eval_samples_per_second": 0.242,
|
|
"eval_steps_per_second": 0.017,
|
|
"step": 20
|
|
},
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.37222222222222223,
|
|
"completions/max_length": 512.0,
|
|
"completions/max_terminated_length": 503.8,
|
|
"completions/mean_length": 416.08055419921874,
|
|
"completions/mean_terminated_length": 359.40147094726564,
|
|
"completions/min_length": 200.0,
|
|
"completions/min_terminated_length": 200.0,
|
|
"epoch": 0.061025223759153785,
|
|
"frac_reward_zero_std": 0.45555556416511533,
|
|
"grad_norm": 1.0458438396453857,
|
|
"kl": 0.0016924589367893835,
|
|
"learning_rate": 5.853658536585365e-07,
|
|
"loss": 0.0238,
|
|
"num_tokens": 1326559.0,
|
|
"reward": 3.1169445514678955,
|
|
"reward_std": 0.6381011128425598,
|
|
"rewards/correctness_reward_func/mean": 1.6916666984558106,
|
|
"rewards/correctness_reward_func/std": 1.468539333343506,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.42527777552604673,
|
|
"rewards/think_structure_reward_func/std": 0.09658884853124619,
|
|
"step": 25
|
|
},
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.3666666666666667,
|
|
"completions/max_length": 512.0,
|
|
"completions/max_terminated_length": 493.2,
|
|
"completions/mean_length": 405.01112060546876,
|
|
"completions/mean_terminated_length": 344.7901611328125,
|
|
"completions/min_length": 209.4,
|
|
"completions/min_terminated_length": 209.4,
|
|
"epoch": 0.07323026851098453,
|
|
"frac_reward_zero_std": 0.3888888955116272,
|
|
"grad_norm": 1.159165620803833,
|
|
"kl": 0.004235438113876929,
|
|
"learning_rate": 7.073170731707316e-07,
|
|
"loss": 0.0255,
|
|
"num_tokens": 1585867.0,
|
|
"reward": 3.1419445037841798,
|
|
"reward_std": 0.7614075779914856,
|
|
"rewards/correctness_reward_func/mean": 1.7166666507720947,
|
|
"rewards/correctness_reward_func/std": 1.439547061920166,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.4252777874469757,
|
|
"rewards/think_structure_reward_func/std": 0.09698245972394944,
|
|
"step": 30
|
|
},
|
|
{
|
|
"epoch": 0.07323026851098453,
|
|
"eval_clip_ratio": 0.0,
|
|
"eval_completions/clipped_ratio": 0.4175,
|
|
"eval_completions/max_length": 512.0,
|
|
"eval_completions/max_terminated_length": 464.56,
|
|
"eval_completions/mean_length": 416.1175,
|
|
"eval_completions/mean_terminated_length": 349.1966192626953,
|
|
"eval_completions/min_length": 262.8,
|
|
"eval_completions/min_terminated_length": 262.8,
|
|
"eval_frac_reward_zero_std": 0.44,
|
|
"eval_kl": 0.006893731448799372,
|
|
"eval_loss": 0.03226928785443306,
|
|
"eval_num_tokens": 1585867.0,
|
|
"eval_reward": 2.9394999980926513,
|
|
"eval_reward_std": 0.625152930021286,
|
|
"eval_rewards/correctness_reward_func/mean": 1.5225,
|
|
"eval_rewards/correctness_reward_func/std": 1.4344161367416381,
|
|
"eval_rewards/format_reward_func/mean": 0.5,
|
|
"eval_rewards/format_reward_func/std": 0.0,
|
|
"eval_rewards/int_reward_func/mean": 0.5,
|
|
"eval_rewards/int_reward_func/std": 0.0,
|
|
"eval_rewards/think_structure_reward_func/mean": 0.41700000286102296,
|
|
"eval_rewards/think_structure_reward_func/std": 0.09943573504686355,
|
|
"eval_runtime": 417.2795,
|
|
"eval_samples_per_second": 0.24,
|
|
"eval_steps_per_second": 0.017,
|
|
"step": 30
|
|
},
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.3833333333333333,
|
|
"completions/max_length": 512.0,
|
|
"completions/max_terminated_length": 501.4,
|
|
"completions/mean_length": 402.71111450195315,
|
|
"completions/mean_terminated_length": 336.4071105957031,
|
|
"completions/min_length": 189.6,
|
|
"completions/min_terminated_length": 189.6,
|
|
"epoch": 0.0854353132628153,
|
|
"frac_reward_zero_std": 0.5000000059604645,
|
|
"grad_norm": 1.128796100616455,
|
|
"kl": 0.011153833094673852,
|
|
"learning_rate": 8.292682926829268e-07,
|
|
"loss": 0.0202,
|
|
"num_tokens": 1843959.0,
|
|
"reward": 3.1919445037841796,
|
|
"reward_std": 0.6783745467662812,
|
|
"rewards/correctness_reward_func/mean": 1.7666666746139525,
|
|
"rewards/correctness_reward_func/std": 1.4569910764694214,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.42527778148651124,
|
|
"rewards/think_structure_reward_func/std": 0.09696891754865647,
|
|
"step": 35
|
|
},
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.20833333333333334,
|
|
"completions/max_length": 512.0,
|
|
"completions/max_terminated_length": 503.6,
|
|
"completions/mean_length": 374.7416625976563,
|
|
"completions/mean_terminated_length": 338.16181640625,
|
|
"completions/min_length": 199.8,
|
|
"completions/min_terminated_length": 199.8,
|
|
"epoch": 0.09764035801464606,
|
|
"frac_reward_zero_std": 0.522222226858139,
|
|
"grad_norm": 1.2832748889923096,
|
|
"kl": 0.03198021717059116,
|
|
"learning_rate": 9.512195121951218e-07,
|
|
"loss": 0.0191,
|
|
"num_tokens": 2092746.0,
|
|
"reward": 3.6022222995758058,
|
|
"reward_std": 0.6069917976856232,
|
|
"rewards/correctness_reward_func/mean": 2.1416666507720947,
|
|
"rewards/correctness_reward_func/std": 1.36202130317688,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.46055554151535033,
|
|
"rewards/think_structure_reward_func/std": 0.07851822599768639,
|
|
"step": 40
|
|
},
|
|
{
|
|
"epoch": 0.09764035801464606,
|
|
"eval_clip_ratio": 0.0,
|
|
"eval_completions/clipped_ratio": 0.31,
|
|
"eval_completions/max_length": 506.32,
|
|
"eval_completions/max_terminated_length": 465.12,
|
|
"eval_completions/mean_length": 398.3625,
|
|
"eval_completions/mean_terminated_length": 348.3086242675781,
|
|
"eval_completions/min_length": 245.48,
|
|
"eval_completions/min_terminated_length": 245.48,
|
|
"eval_frac_reward_zero_std": 0.54,
|
|
"eval_kl": 0.04722412109375,
|
|
"eval_loss": 0.029347257688641548,
|
|
"eval_num_tokens": 2092746.0,
|
|
"eval_reward": 3.0742499923706053,
|
|
"eval_reward_std": 0.49817857772111895,
|
|
"eval_rewards/correctness_reward_func/mean": 1.635,
|
|
"eval_rewards/correctness_reward_func/std": 1.3855872583389282,
|
|
"eval_rewards/format_reward_func/mean": 0.5,
|
|
"eval_rewards/format_reward_func/std": 0.0,
|
|
"eval_rewards/int_reward_func/mean": 0.5,
|
|
"eval_rewards/int_reward_func/std": 0.0,
|
|
"eval_rewards/think_structure_reward_func/mean": 0.4392500054836273,
|
|
"eval_rewards/think_structure_reward_func/std": 0.08431644231081009,
|
|
"eval_runtime": 411.1881,
|
|
"eval_samples_per_second": 0.243,
|
|
"eval_steps_per_second": 0.017,
|
|
"step": 40
|
|
},
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.20833333333333334,
|
|
"completions/max_length": 512.0,
|
|
"completions/max_terminated_length": 503.4,
|
|
"completions/mean_length": 365.9055480957031,
|
|
"completions/mean_terminated_length": 326.92662353515624,
|
|
"completions/min_length": 157.8,
|
|
"completions/min_terminated_length": 157.8,
|
|
"epoch": 0.10984540276647681,
|
|
"frac_reward_zero_std": 0.5111111223697662,
|
|
"grad_norm": 1.1141313314437866,
|
|
"kl": 0.06476407044877609,
|
|
"learning_rate": 9.99836918040428e-07,
|
|
"loss": 0.0079,
|
|
"num_tokens": 2337256.0,
|
|
"reward": 3.4086110591888428,
|
|
"reward_std": 0.6632762014865875,
|
|
"rewards/correctness_reward_func/mean": 1.95,
|
|
"rewards/correctness_reward_func/std": 1.4105015039443969,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.45861111879348754,
|
|
"rewards/think_structure_reward_func/std": 0.08195775151252746,
|
|
"step": 45
|
|
},
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.25277777777777777,
|
|
"completions/max_length": 512.0,
|
|
"completions/max_terminated_length": 491.0,
|
|
"completions/mean_length": 382.7916625976562,
|
|
"completions/mean_terminated_length": 339.6951477050781,
|
|
"completions/min_length": 198.2,
|
|
"completions/min_terminated_length": 198.2,
|
|
"epoch": 0.12205044751830757,
|
|
"frac_reward_zero_std": 0.4888888955116272,
|
|
"grad_norm": 1.404288649559021,
|
|
"kl": 0.08446078604708115,
|
|
"learning_rate": 9.988406912941589e-07,
|
|
"loss": 0.0263,
|
|
"num_tokens": 2588901.0,
|
|
"reward": 3.417500066757202,
|
|
"reward_std": 0.5775101840496063,
|
|
"rewards/correctness_reward_func/mean": 1.9666666746139527,
|
|
"rewards/correctness_reward_func/std": 1.4306164264678956,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.4508333384990692,
|
|
"rewards/think_structure_reward_func/std": 0.08622402399778366,
|
|
"step": 50
|
|
},
|
|
{
|
|
"epoch": 0.12205044751830757,
|
|
"eval_clip_ratio": 0.0,
|
|
"eval_completions/clipped_ratio": 0.295,
|
|
"eval_completions/max_length": 501.92,
|
|
"eval_completions/max_terminated_length": 459.56,
|
|
"eval_completions/mean_length": 393.2975,
|
|
"eval_completions/mean_terminated_length": 346.314892578125,
|
|
"eval_completions/min_length": 237.12,
|
|
"eval_completions/min_terminated_length": 237.12,
|
|
"eval_frac_reward_zero_std": 0.51,
|
|
"eval_kl": 0.08666556805372239,
|
|
"eval_loss": 0.023148151114583015,
|
|
"eval_num_tokens": 2588901.0,
|
|
"eval_reward": 3.098499994277954,
|
|
"eval_reward_std": 0.5522409979999066,
|
|
"eval_rewards/correctness_reward_func/mean": 1.6575,
|
|
"eval_rewards/correctness_reward_func/std": 1.368493332862854,
|
|
"eval_rewards/format_reward_func/mean": 0.5,
|
|
"eval_rewards/format_reward_func/std": 0.0,
|
|
"eval_rewards/int_reward_func/mean": 0.5,
|
|
"eval_rewards/int_reward_func/std": 0.0,
|
|
"eval_rewards/think_structure_reward_func/mean": 0.4409999978542328,
|
|
"eval_rewards/think_structure_reward_func/std": 0.08531437858939171,
|
|
"eval_runtime": 406.1979,
|
|
"eval_samples_per_second": 0.246,
|
|
"eval_steps_per_second": 0.017,
|
|
"step": 50
|
|
},
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.3,
|
|
"completions/max_length": 512.0,
|
|
"completions/max_terminated_length": 502.6,
|
|
"completions/mean_length": 390.2583374023437,
|
|
"completions/mean_terminated_length": 338.8341979980469,
|
|
"completions/min_length": 207.6,
|
|
"completions/min_terminated_length": 207.6,
|
|
"epoch": 0.13425549227013833,
|
|
"frac_reward_zero_std": 0.4222222328186035,
|
|
"grad_norm": 1.2812987565994263,
|
|
"kl": 0.11830147790412109,
|
|
"learning_rate": 9.969406417112488e-07,
|
|
"loss": 0.0236,
|
|
"num_tokens": 2843578.0,
|
|
"reward": 3.1247222423553467,
|
|
"reward_std": 0.6863292157649994,
|
|
"rewards/correctness_reward_func/mean": 1.6833333253860474,
|
|
"rewards/correctness_reward_func/std": 1.4761980772018433,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.44138888120651243,
|
|
"rewards/think_structure_reward_func/std": 0.09145770221948624,
|
|
"step": 55
|
|
},
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.24722222222222223,
|
|
"completions/max_length": 512.0,
|
|
"completions/max_terminated_length": 497.2,
|
|
"completions/mean_length": 383.45556640625,
|
|
"completions/mean_terminated_length": 341.4122314453125,
|
|
"completions/min_length": 172.4,
|
|
"completions/min_terminated_length": 172.4,
|
|
"epoch": 0.14646053702196907,
|
|
"frac_reward_zero_std": 0.522222226858139,
|
|
"grad_norm": 1.2153165340423584,
|
|
"kl": 0.17738193335632482,
|
|
"learning_rate": 9.941402118901742e-07,
|
|
"loss": 0.008,
|
|
"num_tokens": 3095434.0,
|
|
"reward": 3.1858334064483644,
|
|
"reward_std": 0.5761965811252594,
|
|
"rewards/correctness_reward_func/mean": 1.7333333253860475,
|
|
"rewards/correctness_reward_func/std": 1.453945279121399,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.4524999916553497,
|
|
"rewards/think_structure_reward_func/std": 0.08397756516933441,
|
|
"step": 60
|
|
},
|
|
{
|
|
"epoch": 0.14646053702196907,
|
|
"eval_clip_ratio": 0.0,
|
|
"eval_completions/clipped_ratio": 0.23,
|
|
"eval_completions/max_length": 505.16,
|
|
"eval_completions/max_terminated_length": 457.08,
|
|
"eval_completions/mean_length": 378.41,
|
|
"eval_completions/mean_terminated_length": 338.3927966308594,
|
|
"eval_completions/min_length": 236.76,
|
|
"eval_completions/min_terminated_length": 236.76,
|
|
"eval_frac_reward_zero_std": 0.52,
|
|
"eval_kl": 0.3469104462862015,
|
|
"eval_loss": 0.031242886558175087,
|
|
"eval_num_tokens": 3095434.0,
|
|
"eval_reward": 3.0394999885559084,
|
|
"eval_reward_std": 0.5152464130520821,
|
|
"eval_rewards/correctness_reward_func/mean": 1.5825,
|
|
"eval_rewards/correctness_reward_func/std": 1.3812697839736938,
|
|
"eval_rewards/format_reward_func/mean": 0.5,
|
|
"eval_rewards/format_reward_func/std": 0.0,
|
|
"eval_rewards/int_reward_func/mean": 0.5,
|
|
"eval_rewards/int_reward_func/std": 0.0,
|
|
"eval_rewards/think_structure_reward_func/mean": 0.4569999992847443,
|
|
"eval_rewards/think_structure_reward_func/std": 0.07454989522695542,
|
|
"eval_runtime": 411.385,
|
|
"eval_samples_per_second": 0.243,
|
|
"eval_steps_per_second": 0.017,
|
|
"step": 60
|
|
},
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.20833333333333334,
|
|
"completions/max_length": 512.0,
|
|
"completions/max_terminated_length": 497.2,
|
|
"completions/mean_length": 373.71111450195315,
|
|
"completions/mean_terminated_length": 337.03223876953126,
|
|
"completions/min_length": 175.6,
|
|
"completions/min_terminated_length": 175.6,
|
|
"epoch": 0.15866558177379983,
|
|
"frac_reward_zero_std": 0.6444444537162781,
|
|
"grad_norm": 1.2431269884109497,
|
|
"kl": 0.5710139732807875,
|
|
"learning_rate": 9.90444475780332e-07,
|
|
"loss": 0.0219,
|
|
"num_tokens": 3344282.0,
|
|
"reward": 3.3194445610046386,
|
|
"reward_std": 0.4152176558971405,
|
|
"rewards/correctness_reward_func/mean": 1.8583333253860475,
|
|
"rewards/correctness_reward_func/std": 1.4419438600540162,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.46111112236976626,
|
|
"rewards/think_structure_reward_func/std": 0.07665464207530022,
|
|
"step": 65
|
|
},
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.19722222222222222,
|
|
"completions/max_length": 512.0,
|
|
"completions/max_terminated_length": 499.2,
|
|
"completions/mean_length": 372.4166687011719,
|
|
"completions/mean_terminated_length": 338.20623779296875,
|
|
"completions/min_length": 181.4,
|
|
"completions/min_terminated_length": 181.4,
|
|
"epoch": 0.1708706265256306,
|
|
"frac_reward_zero_std": 0.5888888955116272,
|
|
"grad_norm": 1.627001166343689,
|
|
"kl": 0.6150037478655577,
|
|
"learning_rate": 9.85860129488821e-07,
|
|
"loss": 0.0246,
|
|
"num_tokens": 3593872.0,
|
|
"reward": 3.2108334064483643,
|
|
"reward_std": 0.41754000782966616,
|
|
"rewards/correctness_reward_func/mean": 1.75,
|
|
"rewards/correctness_reward_func/std": 1.4730451345443725,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.4608333110809326,
|
|
"rewards/think_structure_reward_func/std": 0.08064375221729278,
|
|
"step": 70
|
|
},
|
|
{
|
|
"epoch": 0.1708706265256306,
|
|
"eval_clip_ratio": 0.0,
|
|
"eval_completions/clipped_ratio": 0.1575,
|
|
"eval_completions/max_length": 496.28,
|
|
"eval_completions/max_terminated_length": 450.08,
|
|
"eval_completions/mean_length": 356.04,
|
|
"eval_completions/mean_terminated_length": 327.35150390625,
|
|
"eval_completions/min_length": 225.16,
|
|
"eval_completions/min_terminated_length": 225.16,
|
|
"eval_frac_reward_zero_std": 0.49,
|
|
"eval_kl": 2.5761199587583543,
|
|
"eval_loss": 0.02504223957657814,
|
|
"eval_num_tokens": 3593872.0,
|
|
"eval_reward": 3.112749991416931,
|
|
"eval_reward_std": 0.5255599319934845,
|
|
"eval_rewards/correctness_reward_func/mean": 1.6425,
|
|
"eval_rewards/correctness_reward_func/std": 1.3481458187103272,
|
|
"eval_rewards/format_reward_func/mean": 0.5,
|
|
"eval_rewards/format_reward_func/std": 0.0,
|
|
"eval_rewards/int_reward_func/mean": 0.5,
|
|
"eval_rewards/int_reward_func/std": 0.0,
|
|
"eval_rewards/think_structure_reward_func/mean": 0.47024999737739565,
|
|
"eval_rewards/think_structure_reward_func/std": 0.06251243278384208,
|
|
"eval_runtime": 419.5979,
|
|
"eval_samples_per_second": 0.238,
|
|
"eval_steps_per_second": 0.017,
|
|
"step": 70
|
|
},
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.17222222222222222,
|
|
"completions/max_length": 512.0,
|
|
"completions/max_terminated_length": 496.2,
|
|
"completions/mean_length": 355.4861145019531,
|
|
"completions/mean_terminated_length": 322.54615478515626,
|
|
"completions/min_length": 170.2,
|
|
"completions/min_terminated_length": 170.2,
|
|
"epoch": 0.18307567127746135,
|
|
"frac_reward_zero_std": 0.5333333432674408,
|
|
"grad_norm": 1.6791362762451172,
|
|
"kl": 0.8810447674244642,
|
|
"learning_rate": 9.803954791481238e-07,
|
|
"loss": 0.0352,
|
|
"num_tokens": 3834695.0,
|
|
"reward": 3.325000047683716,
|
|
"reward_std": 0.5272651016712189,
|
|
"rewards/correctness_reward_func/mean": 1.8583333253860475,
|
|
"rewards/correctness_reward_func/std": 1.4408273458480836,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.4666666805744171,
|
|
"rewards/think_structure_reward_func/std": 0.07259104251861573,
|
|
"step": 75
|
|
},
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.1222222222222222,
|
|
"completions/max_length": 509.2,
|
|
"completions/max_terminated_length": 496.8,
|
|
"completions/mean_length": 333.54722290039064,
|
|
"completions/mean_terminated_length": 308.826904296875,
|
|
"completions/min_length": 160.0,
|
|
"completions/min_terminated_length": 160.0,
|
|
"epoch": 0.19528071602929212,
|
|
"frac_reward_zero_std": 0.6333333313465118,
|
|
"grad_norm": 0.818752110004425,
|
|
"kl": 1.7537097098926704,
|
|
"learning_rate": 9.740604258666668e-07,
|
|
"loss": 0.0168,
|
|
"num_tokens": 4068248.0,
|
|
"reward": 3.42722225189209,
|
|
"reward_std": 0.5413465321063995,
|
|
"rewards/correctness_reward_func/mean": 1.95,
|
|
"rewards/correctness_reward_func/std": 1.4035101175308227,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.4772222101688385,
|
|
"rewards/think_structure_reward_func/std": 0.05554628893733025,
|
|
"step": 80
|
|
},
|
|
{
|
|
"epoch": 0.19528071602929212,
|
|
"eval_clip_ratio": 0.0,
|
|
"eval_completions/clipped_ratio": 0.1125,
|
|
"eval_completions/max_length": 482.12,
|
|
"eval_completions/max_terminated_length": 437.44,
|
|
"eval_completions/mean_length": 337.9975,
|
|
"eval_completions/mean_terminated_length": 316.35515686035154,
|
|
"eval_completions/min_length": 214.48,
|
|
"eval_completions/min_terminated_length": 214.48,
|
|
"eval_frac_reward_zero_std": 0.51,
|
|
"eval_kl": 7.1868480104208,
|
|
"eval_loss": 0.022400854155421257,
|
|
"eval_num_tokens": 4068248.0,
|
|
"eval_reward": 3.0174999904632567,
|
|
"eval_reward_std": 0.5413145567476749,
|
|
"eval_rewards/correctness_reward_func/mean": 1.5375,
|
|
"eval_rewards/correctness_reward_func/std": 1.3865587854385375,
|
|
"eval_rewards/format_reward_func/mean": 0.5,
|
|
"eval_rewards/format_reward_func/std": 0.0,
|
|
"eval_rewards/int_reward_func/mean": 0.5,
|
|
"eval_rewards/int_reward_func/std": 0.0,
|
|
"eval_rewards/think_structure_reward_func/mean": 0.48,
|
|
"eval_rewards/think_structure_reward_func/std": 0.04754733845591545,
|
|
"eval_runtime": 410.4211,
|
|
"eval_samples_per_second": 0.244,
|
|
"eval_steps_per_second": 0.017,
|
|
"step": 80
|
|
},
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.10833333333333335,
|
|
"completions/max_length": 512.0,
|
|
"completions/max_terminated_length": 495.2,
|
|
"completions/mean_length": 343.2388916015625,
|
|
"completions/mean_terminated_length": 322.68097534179685,
|
|
"completions/min_length": 150.4,
|
|
"completions/min_terminated_length": 150.4,
|
|
"epoch": 0.20748576078112285,
|
|
"frac_reward_zero_std": 0.6333333373069763,
|
|
"grad_norm": 1.471542239189148,
|
|
"kl": 1.2616392819831768,
|
|
"learning_rate": 9.66866447789531e-07,
|
|
"loss": 0.0149,
|
|
"num_tokens": 4306326.0,
|
|
"reward": 3.28666672706604,
|
|
"reward_std": 0.4829041063785553,
|
|
"rewards/correctness_reward_func/mean": 1.8083333253860474,
|
|
"rewards/correctness_reward_func/std": 1.4639185905456542,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.47833333611488343,
|
|
"rewards/think_structure_reward_func/std": 0.060376092046499255,
|
|
"step": 85
|
|
},
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.0916666666666667,
|
|
"completions/max_length": 512.0,
|
|
"completions/max_terminated_length": 496.8,
|
|
"completions/mean_length": 329.2805541992187,
|
|
"completions/mean_terminated_length": 311.8640502929687,
|
|
"completions/min_length": 140.4,
|
|
"completions/min_terminated_length": 140.4,
|
|
"epoch": 0.21969080553295361,
|
|
"frac_reward_zero_std": 0.5444444596767426,
|
|
"grad_norm": 1.6604244709014893,
|
|
"kl": 0.9448512117067973,
|
|
"learning_rate": 9.58826579301814e-07,
|
|
"loss": 0.027,
|
|
"num_tokens": 4538771.0,
|
|
"reward": 3.2663888931274414,
|
|
"reward_std": 0.500895208120346,
|
|
"rewards/correctness_reward_func/mean": 1.7833333492279053,
|
|
"rewards/correctness_reward_func/std": 1.4695966958999633,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.48305554389953614,
|
|
"rewards/think_structure_reward_func/std": 0.05385549664497376,
|
|
"step": 90
|
|
},
|
|
{
|
|
"epoch": 0.21969080553295361,
|
|
"eval_clip_ratio": 0.0,
|
|
"eval_completions/clipped_ratio": 0.1325,
|
|
"eval_completions/max_length": 481.52,
|
|
"eval_completions/max_terminated_length": 434.8,
|
|
"eval_completions/mean_length": 338.5575,
|
|
"eval_completions/mean_terminated_length": 311.81354736328126,
|
|
"eval_completions/min_length": 209.36,
|
|
"eval_completions/min_terminated_length": 209.36,
|
|
"eval_frac_reward_zero_std": 0.5,
|
|
"eval_kl": 1.148498348593712,
|
|
"eval_loss": 0.028217840939760208,
|
|
"eval_num_tokens": 4538771.0,
|
|
"eval_reward": 3.033499994277954,
|
|
"eval_reward_std": 0.6048994866013527,
|
|
"eval_rewards/correctness_reward_func/mean": 1.56,
|
|
"eval_rewards/correctness_reward_func/std": 1.3664008426666259,
|
|
"eval_rewards/format_reward_func/mean": 0.5,
|
|
"eval_rewards/format_reward_func/std": 0.0,
|
|
"eval_rewards/int_reward_func/mean": 0.5,
|
|
"eval_rewards/int_reward_func/std": 0.0,
|
|
"eval_rewards/think_structure_reward_func/mean": 0.4734999978542328,
|
|
"eval_rewards/think_structure_reward_func/std": 0.052927846163511275,
|
|
"eval_runtime": 406.6898,
|
|
"eval_samples_per_second": 0.246,
|
|
"eval_steps_per_second": 0.017,
|
|
"step": 90
|
|
},
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.09722222222222221,
|
|
"completions/max_length": 512.0,
|
|
"completions/max_terminated_length": 484.0,
|
|
"completions/mean_length": 329.7805541992187,
|
|
"completions/mean_terminated_length": 310.32548828125,
|
|
"completions/min_length": 151.4,
|
|
"completions/min_terminated_length": 151.4,
|
|
"epoch": 0.23189585028478438,
|
|
"frac_reward_zero_std": 0.5666666805744172,
|
|
"grad_norm": 1.5117992162704468,
|
|
"kl": 1.0005333493153254,
|
|
"learning_rate": 9.499553874123212e-07,
|
|
"loss": 0.0078,
|
|
"num_tokens": 4770880.0,
|
|
"reward": 3.2063889026641847,
|
|
"reward_std": 0.5465041697025299,
|
|
"rewards/correctness_reward_func/mean": 1.725,
|
|
"rewards/correctness_reward_func/std": 1.4602257013320923,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.4813888847827911,
|
|
"rewards/think_structure_reward_func/std": 0.05693056583404541,
|
|
"step": 95
|
|
},
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.08888888888888888,
|
|
"completions/max_length": 506.4,
|
|
"completions/max_terminated_length": 495.2,
|
|
"completions/mean_length": 319.5388977050781,
|
|
"completions/mean_terminated_length": 300.96067504882814,
|
|
"completions/min_length": 154.2,
|
|
"completions/min_terminated_length": 154.2,
|
|
"epoch": 0.24410089503661514,
|
|
"frac_reward_zero_std": 0.5111111164093017,
|
|
"grad_norm": 2.2661139965057373,
|
|
"kl": 1.6817767086128395,
|
|
"learning_rate": 9.402689453603814e-07,
|
|
"loss": 0.0069,
|
|
"num_tokens": 4999866.0,
|
|
"reward": 3.2255555629730224,
|
|
"reward_std": 0.7024516820907593,
|
|
"rewards/correctness_reward_func/mean": 1.7416666746139526,
|
|
"rewards/correctness_reward_func/std": 1.4767550468444823,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.48388888239860534,
|
|
"rewards/think_structure_reward_func/std": 0.04678683057427406,
|
|
"step": 100
|
|
},
|
|
{
|
|
"epoch": 0.24410089503661514,
|
|
"eval_clip_ratio": 0.0,
|
|
"eval_completions/clipped_ratio": 0.095,
|
|
"eval_completions/max_length": 482.44,
|
|
"eval_completions/max_terminated_length": 438.64,
|
|
"eval_completions/mean_length": 325.12,
|
|
"eval_completions/mean_terminated_length": 306.41920776367186,
|
|
"eval_completions/min_length": 206.68,
|
|
"eval_completions/min_terminated_length": 206.68,
|
|
"eval_frac_reward_zero_std": 0.47,
|
|
"eval_kl": 0.9050960391759872,
|
|
"eval_loss": 0.031647272408008575,
|
|
"eval_num_tokens": 4999866.0,
|
|
"eval_reward": 3.131499967575073,
|
|
"eval_reward_std": 0.7053796461224556,
|
|
"eval_rewards/correctness_reward_func/mean": 1.65,
|
|
"eval_rewards/correctness_reward_func/std": 1.4236406707763671,
|
|
"eval_rewards/format_reward_func/mean": 0.5,
|
|
"eval_rewards/format_reward_func/std": 0.0,
|
|
"eval_rewards/int_reward_func/mean": 0.5,
|
|
"eval_rewards/int_reward_func/std": 0.0,
|
|
"eval_rewards/think_structure_reward_func/mean": 0.48149999856948855,
|
|
"eval_rewards/think_structure_reward_func/std": 0.046209767013788226,
|
|
"eval_runtime": 409.718,
|
|
"eval_samples_per_second": 0.244,
|
|
"eval_steps_per_second": 0.017,
|
|
"step": 100
|
|
},
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.08055555555555556,
|
|
"completions/max_length": 512.0,
|
|
"completions/max_terminated_length": 500.8,
|
|
"completions/mean_length": 331.802783203125,
|
|
"completions/mean_terminated_length": 315.9258178710937,
|
|
"completions/min_length": 163.8,
|
|
"completions/min_terminated_length": 163.8,
|
|
"epoch": 0.2563059397884459,
|
|
"frac_reward_zero_std": 0.5444444596767426,
|
|
"grad_norm": 1.2682464122772217,
|
|
"kl": 1.8855398039023081,
|
|
"learning_rate": 9.297848034936005e-07,
|
|
"loss": 0.0153,
|
|
"num_tokens": 5234811.0,
|
|
"reward": 3.160555601119995,
|
|
"reward_std": 0.556890806555748,
|
|
"rewards/correctness_reward_func/mean": 1.6750000238418579,
|
|
"rewards/correctness_reward_func/std": 1.4766180038452148,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.4855555593967438,
|
|
"rewards/think_structure_reward_func/std": 0.05022132247686386,
|
|
"step": 105
|
|
},
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.06944444444444446,
|
|
"completions/max_length": 512.0,
|
|
"completions/max_terminated_length": 497.6,
|
|
"completions/mean_length": 318.03333740234376,
|
|
"completions/mean_terminated_length": 303.60321044921875,
|
|
"completions/min_length": 142.8,
|
|
"completions/min_terminated_length": 142.8,
|
|
"epoch": 0.26851098454027666,
|
|
"frac_reward_zero_std": 0.5333333432674408,
|
|
"grad_norm": 1.5059627294540405,
|
|
"kl": 1.0062146712094546,
|
|
"learning_rate": 9.185219574693241e-07,
|
|
"loss": 0.0016,
|
|
"num_tokens": 5462999.0,
|
|
"reward": 3.4027778625488283,
|
|
"reward_std": 0.6819790482521058,
|
|
"rewards/correctness_reward_func/mean": 1.9166666746139527,
|
|
"rewards/correctness_reward_func/std": 1.4348475933074951,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.48611109852790835,
|
|
"rewards/think_structure_reward_func/std": 0.048446450382471085,
|
|
"step": 110
|
|
},
|
|
{
|
|
"epoch": 0.26851098454027666,
|
|
"eval_clip_ratio": 0.0,
|
|
"eval_completions/clipped_ratio": 0.095,
|
|
"eval_completions/max_length": 482.36,
|
|
"eval_completions/max_terminated_length": 447.16,
|
|
"eval_completions/mean_length": 339.8625,
|
|
"eval_completions/mean_terminated_length": 322.56604248046875,
|
|
"eval_completions/min_length": 216.04,
|
|
"eval_completions/min_terminated_length": 216.04,
|
|
"eval_frac_reward_zero_std": 0.56,
|
|
"eval_kl": 1.3058723711967468,
|
|
"eval_loss": 0.02592286840081215,
|
|
"eval_num_tokens": 5462999.0,
|
|
"eval_reward": 3.1625000190734864,
|
|
"eval_reward_std": 0.5343751847743988,
|
|
"eval_rewards/correctness_reward_func/mean": 1.68,
|
|
"eval_rewards/correctness_reward_func/std": 1.318399386405945,
|
|
"eval_rewards/format_reward_func/mean": 0.5,
|
|
"eval_rewards/format_reward_func/std": 0.0,
|
|
"eval_rewards/int_reward_func/mean": 0.5,
|
|
"eval_rewards/int_reward_func/std": 0.0,
|
|
"eval_rewards/think_structure_reward_func/mean": 0.4825000011920929,
|
|
"eval_rewards/think_structure_reward_func/std": 0.039322869181633,
|
|
"eval_runtime": 408.5023,
|
|
"eval_samples_per_second": 0.245,
|
|
"eval_steps_per_second": 0.017,
|
|
"step": 110
|
|
},
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.12777777777777777,
|
|
"completions/max_length": 512.0,
|
|
"completions/max_terminated_length": 499.2,
|
|
"completions/mean_length": 335.9888977050781,
|
|
"completions/mean_terminated_length": 310.49241333007814,
|
|
"completions/min_length": 158.0,
|
|
"completions/min_terminated_length": 158.0,
|
|
"epoch": 0.2807160292921074,
|
|
"frac_reward_zero_std": 0.5333333373069763,
|
|
"grad_norm": 63.426876068115234,
|
|
"kl": 2.0912602316588162,
|
|
"learning_rate": 9.065008138374188e-07,
|
|
"loss": 0.0154,
|
|
"num_tokens": 5697295.0,
|
|
"reward": 3.4755556106567385,
|
|
"reward_std": 0.6123797535896301,
|
|
"rewards/correctness_reward_func/mean": 1.9999999761581422,
|
|
"rewards/correctness_reward_func/std": 1.4048724889755249,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.47555556893348694,
|
|
"rewards/think_structure_reward_func/std": 0.06329894065856934,
|
|
"step": 115
|
|
},
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.12222222222222226,
|
|
"completions/max_length": 512.0,
|
|
"completions/max_terminated_length": 499.4,
|
|
"completions/mean_length": 346.5305541992187,
|
|
"completions/mean_terminated_length": 323.63262939453125,
|
|
"completions/min_length": 170.0,
|
|
"completions/min_terminated_length": 170.0,
|
|
"epoch": 0.29292107404393813,
|
|
"frac_reward_zero_std": 0.5444444596767426,
|
|
"grad_norm": 1.581868290901184,
|
|
"kl": 1.3886217889686425,
|
|
"learning_rate": 8.937431530667327e-07,
|
|
"loss": 0.0129,
|
|
"num_tokens": 5935986.0,
|
|
"reward": 3.0361112117767335,
|
|
"reward_std": 0.5907206833362579,
|
|
"rewards/correctness_reward_func/mean": 1.5583333730697633,
|
|
"rewards/correctness_reward_func/std": 1.4837919235229493,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.47777777910232544,
|
|
"rewards/think_structure_reward_func/std": 0.06253288313746452,
|
|
"step": 120
|
|
},
|
|
{
|
|
"epoch": 0.29292107404393813,
|
|
"eval_clip_ratio": 0.0,
|
|
"eval_completions/clipped_ratio": 0.1,
|
|
"eval_completions/max_length": 477.56,
|
|
"eval_completions/max_terminated_length": 444.04,
|
|
"eval_completions/mean_length": 339.91,
|
|
"eval_completions/mean_terminated_length": 321.73902893066406,
|
|
"eval_completions/min_length": 217.12,
|
|
"eval_completions/min_terminated_length": 217.12,
|
|
"eval_frac_reward_zero_std": 0.5,
|
|
"eval_kl": 0.9355611637234688,
|
|
"eval_loss": 0.028858236968517303,
|
|
"eval_num_tokens": 5935986.0,
|
|
"eval_reward": 3.123499984741211,
|
|
"eval_reward_std": 0.5698655286431312,
|
|
"eval_rewards/correctness_reward_func/mean": 1.6425,
|
|
"eval_rewards/correctness_reward_func/std": 1.2987245893478394,
|
|
"eval_rewards/format_reward_func/mean": 0.5,
|
|
"eval_rewards/format_reward_func/std": 0.0,
|
|
"eval_rewards/int_reward_func/mean": 0.5,
|
|
"eval_rewards/int_reward_func/std": 0.0,
|
|
"eval_rewards/think_structure_reward_func/mean": 0.4809999990463257,
|
|
"eval_rewards/think_structure_reward_func/std": 0.0445744352042675,
|
|
"eval_runtime": 400.3071,
|
|
"eval_samples_per_second": 0.25,
|
|
"eval_steps_per_second": 0.017,
|
|
"step": 120
|
|
},
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.08333333333333333,
|
|
"completions/max_length": 512.0,
|
|
"completions/max_terminated_length": 493.0,
|
|
"completions/mean_length": 341.44166870117186,
|
|
"completions/mean_terminated_length": 326.6929077148437,
|
|
"completions/min_length": 178.8,
|
|
"completions/min_terminated_length": 178.8,
|
|
"epoch": 0.3051261187957689,
|
|
"frac_reward_zero_std": 0.6444444537162781,
|
|
"grad_norm": 0.911089301109314,
|
|
"kl": 0.4451469591508309,
|
|
"learning_rate": 8.802720900822269e-07,
|
|
"loss": 0.0136,
|
|
"num_tokens": 6174497.0,
|
|
"reward": 3.6261112689971924,
|
|
"reward_std": 0.46896895170211794,
|
|
"rewards/correctness_reward_func/mean": 2.1416666507720947,
|
|
"rewards/correctness_reward_func/std": 1.3475727796554566,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.4844444334506989,
|
|
"rewards/think_structure_reward_func/std": 0.05166236087679863,
|
|
"step": 125
|
|
},
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.1,
|
|
"completions/max_length": 512.0,
|
|
"completions/max_terminated_length": 495.8,
|
|
"completions/mean_length": 334.3861083984375,
|
|
"completions/mean_terminated_length": 314.84796142578125,
|
|
"completions/min_length": 153.8,
|
|
"completions/min_terminated_length": 153.8,
|
|
"epoch": 0.31733116354759966,
|
|
"frac_reward_zero_std": 0.6222222208976745,
|
|
"grad_norm": 4.220540523529053,
|
|
"kl": 2.1644184393187365,
|
|
"learning_rate": 8.66112032384275e-07,
|
|
"loss": 0.0218,
|
|
"num_tokens": 6409332.0,
|
|
"reward": 3.0972222328186034,
|
|
"reward_std": 0.49864366054534914,
|
|
"rewards/correctness_reward_func/mean": 1.6166666746139526,
|
|
"rewards/correctness_reward_func/std": 1.4961567401885987,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.4805555582046509,
|
|
"rewards/think_structure_reward_func/std": 0.05694576576352119,
|
|
"step": 130
|
|
},
|
|
{
|
|
"epoch": 0.31733116354759966,
|
|
"eval_clip_ratio": 0.0,
|
|
"eval_completions/clipped_ratio": 0.0775,
|
|
"eval_completions/max_length": 475.6,
|
|
"eval_completions/max_terminated_length": 446.12,
|
|
"eval_completions/mean_length": 329.9925,
|
|
"eval_completions/mean_terminated_length": 314.5217761230469,
|
|
"eval_completions/min_length": 215.88,
|
|
"eval_completions/min_terminated_length": 215.88,
|
|
"eval_frac_reward_zero_std": 0.55,
|
|
"eval_kl": 1.0998025035858154,
|
|
"eval_loss": 0.004488673526793718,
|
|
"eval_num_tokens": 6409332.0,
|
|
"eval_reward": 3.1047499847412108,
|
|
"eval_reward_std": 0.5491062951087952,
|
|
"eval_rewards/correctness_reward_func/mean": 1.62,
|
|
"eval_rewards/correctness_reward_func/std": 1.3408914089202881,
|
|
"eval_rewards/format_reward_func/mean": 0.5,
|
|
"eval_rewards/format_reward_func/std": 0.0,
|
|
"eval_rewards/int_reward_func/mean": 0.5,
|
|
"eval_rewards/int_reward_func/std": 0.0,
|
|
"eval_rewards/think_structure_reward_func/mean": 0.4847500002384186,
|
|
"eval_rewards/think_structure_reward_func/std": 0.03901088729500771,
|
|
"eval_runtime": 396.7856,
|
|
"eval_samples_per_second": 0.252,
|
|
"eval_steps_per_second": 0.018,
|
|
"step": 130
|
|
},
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.061111111111111095,
|
|
"completions/max_length": 512.0,
|
|
"completions/max_terminated_length": 468.4,
|
|
"completions/mean_length": 317.7277893066406,
|
|
"completions/mean_terminated_length": 305.14564819335936,
|
|
"completions/min_length": 152.2,
|
|
"completions/min_terminated_length": 152.2,
|
|
"epoch": 0.3295362082994304,
|
|
"frac_reward_zero_std": 0.6555555582046508,
|
|
"grad_norm": 1.2128106355667114,
|
|
"kl": 0.6293413537244003,
|
|
"learning_rate": 8.51288635826016e-07,
|
|
"loss": 0.0217,
|
|
"num_tokens": 6637402.0,
|
|
"reward": 3.271666669845581,
|
|
"reward_std": 0.4769443452358246,
|
|
"rewards/correctness_reward_func/mean": 1.7833333015441895,
|
|
"rewards/correctness_reward_func/std": 1.441186261177063,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.48833333849906924,
|
|
"rewards/think_structure_reward_func/std": 0.044604333490133284,
|
|
"step": 135
|
|
},
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.061111111111111116,
|
|
"completions/max_length": 512.0,
|
|
"completions/max_terminated_length": 481.8,
|
|
"completions/mean_length": 312.6750061035156,
|
|
"completions/mean_terminated_length": 299.92626953125,
|
|
"completions/min_length": 145.2,
|
|
"completions/min_terminated_length": 145.2,
|
|
"epoch": 0.3417412530512612,
|
|
"frac_reward_zero_std": 0.6444444477558136,
|
|
"grad_norm": 0.8413270115852356,
|
|
"kl": 0.6597472462803126,
|
|
"learning_rate": 8.358287581288822e-07,
|
|
"loss": 0.0078,
|
|
"num_tokens": 6863309.0,
|
|
"reward": 3.505555534362793,
|
|
"reward_std": 0.4724615037441254,
|
|
"rewards/correctness_reward_func/mean": 2.0166666746139525,
|
|
"rewards/correctness_reward_func/std": 1.4095194816589356,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.4888888955116272,
|
|
"rewards/think_structure_reward_func/std": 0.045268140733242035,
|
|
"step": 140
|
|
},
|
|
{
|
|
"epoch": 0.3417412530512612,
|
|
"eval_clip_ratio": 0.0,
|
|
"eval_completions/clipped_ratio": 0.0575,
|
|
"eval_completions/max_length": 452.56,
|
|
"eval_completions/max_terminated_length": 429.8,
|
|
"eval_completions/mean_length": 308.505,
|
|
"eval_completions/mean_terminated_length": 296.7969323730469,
|
|
"eval_completions/min_length": 198.44,
|
|
"eval_completions/min_terminated_length": 198.44,
|
|
"eval_frac_reward_zero_std": 0.57,
|
|
"eval_kl": 249.94064613878726,
|
|
"eval_loss": 0.2721412181854248,
|
|
"eval_num_tokens": 6863309.0,
|
|
"eval_reward": 3.0879999828338622,
|
|
"eval_reward_std": 0.5598350358009339,
|
|
"eval_rewards/correctness_reward_func/mean": 1.5975,
|
|
"eval_rewards/correctness_reward_func/std": 1.3702752017974853,
|
|
"eval_rewards/format_reward_func/mean": 0.5,
|
|
"eval_rewards/format_reward_func/std": 0.0,
|
|
"eval_rewards/int_reward_func/mean": 0.5,
|
|
"eval_rewards/int_reward_func/std": 0.0,
|
|
"eval_rewards/think_structure_reward_func/mean": 0.49050000309944153,
|
|
"eval_rewards/think_structure_reward_func/std": 0.026519651859998702,
|
|
"eval_runtime": 379.3445,
|
|
"eval_samples_per_second": 0.264,
|
|
"eval_steps_per_second": 0.018,
|
|
"step": 140
|
|
},
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.08333333333333333,
|
|
"completions/max_length": 512.0,
|
|
"completions/max_terminated_length": 487.6,
|
|
"completions/mean_length": 314.35834350585935,
|
|
"completions/mean_terminated_length": 296.4543518066406,
|
|
"completions/min_length": 141.6,
|
|
"completions/min_terminated_length": 141.6,
|
|
"epoch": 0.35394629780309195,
|
|
"frac_reward_zero_std": 0.6555555701255799,
|
|
"grad_norm": 1.3788807392120361,
|
|
"kl": 0.9813333005954822,
|
|
"learning_rate": 8.19760410220527e-07,
|
|
"loss": 0.0087,
|
|
"num_tokens": 7091638.0,
|
|
"reward": 3.359444570541382,
|
|
"reward_std": 0.42976476550102233,
|
|
"rewards/correctness_reward_func/mean": 1.8750000238418578,
|
|
"rewards/correctness_reward_func/std": 1.4225042343139649,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.4844444453716278,
|
|
"rewards/think_structure_reward_func/std": 0.0534981407225132,
|
|
"step": 145
|
|
},
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.033333333333333326,
|
|
"completions/max_length": 505.4,
|
|
"completions/max_terminated_length": 492.2,
|
|
"completions/mean_length": 310.52500610351564,
|
|
"completions/mean_terminated_length": 303.3688598632813,
|
|
"completions/min_length": 169.6,
|
|
"completions/min_terminated_length": 169.6,
|
|
"epoch": 0.3661513425549227,
|
|
"frac_reward_zero_std": 0.6444444537162781,
|
|
"grad_norm": 2.2258527278900146,
|
|
"kl": 1.4337066128849982,
|
|
"learning_rate": 8.03112705483319e-07,
|
|
"loss": 0.0179,
|
|
"num_tokens": 7316943.0,
|
|
"reward": 3.3861111640930175,
|
|
"reward_std": 0.503593909740448,
|
|
"rewards/correctness_reward_func/mean": 1.8916666746139525,
|
|
"rewards/correctness_reward_func/std": 1.449527096748352,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.4944444358348846,
|
|
"rewards/think_structure_reward_func/std": 0.02908540740609169,
|
|
"step": 150
|
|
},
|
|
{
|
|
"epoch": 0.3661513425549227,
|
|
"eval_clip_ratio": 0.0,
|
|
"eval_completions/clipped_ratio": 0.0275,
|
|
"eval_completions/max_length": 443.76,
|
|
"eval_completions/max_terminated_length": 427.8,
|
|
"eval_completions/mean_length": 304.0275,
|
|
"eval_completions/mean_terminated_length": 298.327197265625,
|
|
"eval_completions/min_length": 198.28,
|
|
"eval_completions/min_terminated_length": 198.28,
|
|
"eval_frac_reward_zero_std": 0.59,
|
|
"eval_kl": 3.954054220318794,
|
|
"eval_loss": 0.012205589562654495,
|
|
"eval_num_tokens": 7316943.0,
|
|
"eval_reward": 3.1074999952316285,
|
|
"eval_reward_std": 0.6089521622657776,
|
|
"eval_rewards/correctness_reward_func/mean": 1.6125,
|
|
"eval_rewards/correctness_reward_func/std": 1.3101955652236938,
|
|
"eval_rewards/format_reward_func/mean": 0.5,
|
|
"eval_rewards/format_reward_func/std": 0.0,
|
|
"eval_rewards/int_reward_func/mean": 0.5,
|
|
"eval_rewards/int_reward_func/std": 0.0,
|
|
"eval_rewards/think_structure_reward_func/mean": 0.4950000023841858,
|
|
"eval_rewards/think_structure_reward_func/std": 0.01595742329955101,
|
|
"eval_runtime": 373.6037,
|
|
"eval_samples_per_second": 0.268,
|
|
"eval_steps_per_second": 0.019,
|
|
"step": 150
|
|
},
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.027777777777777745,
|
|
"completions/max_length": 512.0,
|
|
"completions/max_terminated_length": 497.0,
|
|
"completions/mean_length": 308.45277709960936,
|
|
"completions/mean_terminated_length": 302.6772705078125,
|
|
"completions/min_length": 149.8,
|
|
"completions/min_terminated_length": 149.8,
|
|
"epoch": 0.37835638730675347,
|
|
"frac_reward_zero_std": 0.6888888955116272,
|
|
"grad_norm": 1.4784741401672363,
|
|
"kl": 1.6666569340974093,
|
|
"learning_rate": 7.859158070053576e-07,
|
|
"loss": 0.0064,
|
|
"num_tokens": 7540914.0,
|
|
"reward": 3.5288889408111572,
|
|
"reward_std": 0.4466042459011078,
|
|
"rewards/correctness_reward_func/mean": 2.0333333730697634,
|
|
"rewards/correctness_reward_func/std": 1.394612693786621,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.4955555498600006,
|
|
"rewards/think_structure_reward_func/std": 0.026002290844917297,
|
|
"step": 155
|
|
},
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.033333333333333326,
|
|
"completions/max_length": 509.2,
|
|
"completions/max_terminated_length": 505.4,
|
|
"completions/mean_length": 293.48333740234375,
|
|
"completions/mean_terminated_length": 286.2913330078125,
|
|
"completions/min_length": 147.6,
|
|
"completions/min_terminated_length": 147.6,
|
|
"epoch": 0.39056143205858423,
|
|
"frac_reward_zero_std": 0.622222238779068,
|
|
"grad_norm": 9.57494831085205,
|
|
"kl": 2.3059448738892874,
|
|
"learning_rate": 7.682008729295833e-07,
|
|
"loss": 0.0143,
|
|
"num_tokens": 7759720.0,
|
|
"reward": 3.234999990463257,
|
|
"reward_std": 0.5269722521305085,
|
|
"rewards/correctness_reward_func/mean": 1.7416666746139526,
|
|
"rewards/correctness_reward_func/std": 1.4664482593536377,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.49333333373069765,
|
|
"rewards/think_structure_reward_func/std": 0.027515591681003572,
|
|
"step": 160
|
|
},
|
|
{
|
|
"epoch": 0.39056143205858423,
|
|
"eval_clip_ratio": 0.0,
|
|
"eval_completions/clipped_ratio": 0.0475,
|
|
"eval_completions/max_length": 457.92,
|
|
"eval_completions/max_terminated_length": 418.64,
|
|
"eval_completions/mean_length": 301.4475,
|
|
"eval_completions/mean_terminated_length": 290.7835424804687,
|
|
"eval_completions/min_length": 193.36,
|
|
"eval_completions/min_terminated_length": 193.36,
|
|
"eval_frac_reward_zero_std": 0.55,
|
|
"eval_kl": 1.0727974826097488,
|
|
"eval_loss": 0.028670670464634895,
|
|
"eval_num_tokens": 7759720.0,
|
|
"eval_reward": 3.0664999866485596,
|
|
"eval_reward_std": 0.6417502629756927,
|
|
"eval_rewards/correctness_reward_func/mean": 1.575,
|
|
"eval_rewards/correctness_reward_func/std": 1.3819786500930786,
|
|
"eval_rewards/format_reward_func/mean": 0.5,
|
|
"eval_rewards/format_reward_func/std": 0.0,
|
|
"eval_rewards/int_reward_func/mean": 0.5,
|
|
"eval_rewards/int_reward_func/std": 0.0,
|
|
"eval_rewards/think_structure_reward_func/mean": 0.4915000033378601,
|
|
"eval_rewards/think_structure_reward_func/std": 0.025535131990909576,
|
|
"eval_runtime": 382.3973,
|
|
"eval_samples_per_second": 0.262,
|
|
"eval_steps_per_second": 0.018,
|
|
"step": 160
|
|
},
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.04166666666666665,
|
|
"completions/max_length": 512.0,
|
|
"completions/max_terminated_length": 498.0,
|
|
"completions/mean_length": 304.15833740234376,
|
|
"completions/mean_terminated_length": 295.2366455078125,
|
|
"completions/min_length": 163.6,
|
|
"completions/min_terminated_length": 163.6,
|
|
"epoch": 0.402766476810415,
|
|
"frac_reward_zero_std": 0.6000000238418579,
|
|
"grad_norm": 1.1254874467849731,
|
|
"kl": 1.1495040642718475,
|
|
"learning_rate": 7.5e-07,
|
|
"loss": 0.0048,
|
|
"num_tokens": 7984237.0,
|
|
"reward": 3.3500000476837157,
|
|
"reward_std": 0.5785813271999359,
|
|
"rewards/correctness_reward_func/mean": 1.8583333015441894,
|
|
"rewards/correctness_reward_func/std": 1.401494526863098,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.49166666269302367,
|
|
"rewards/think_structure_reward_func/std": 0.03884918764233589,
|
|
"step": 165
|
|
},
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.036111111111111115,
|
|
"completions/max_length": 494.4,
|
|
"completions/max_terminated_length": 469.2,
|
|
"completions/mean_length": 295.6972290039063,
|
|
"completions/mean_terminated_length": 287.7515380859375,
|
|
"completions/min_length": 145.2,
|
|
"completions/min_terminated_length": 145.2,
|
|
"epoch": 0.4149715215622457,
|
|
"frac_reward_zero_std": 0.5777777850627899,
|
|
"grad_norm": 13.315486907958984,
|
|
"kl": 2.232921237250169,
|
|
"learning_rate": 7.313461654072973e-07,
|
|
"loss": 0.0149,
|
|
"num_tokens": 8205108.0,
|
|
"reward": 3.434444522857666,
|
|
"reward_std": 0.602297329902649,
|
|
"rewards/correctness_reward_func/mean": 1.9416666507720948,
|
|
"rewards/correctness_reward_func/std": 1.4108514785766602,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.49277777075767515,
|
|
"rewards/think_structure_reward_func/std": 0.027992242574691774,
|
|
"step": 170
|
|
},
|
|
{
|
|
"epoch": 0.4149715215622457,
|
|
"eval_clip_ratio": 0.0,
|
|
"eval_completions/clipped_ratio": 0.055,
|
|
"eval_completions/max_length": 437.16,
|
|
"eval_completions/max_terminated_length": 415.2,
|
|
"eval_completions/mean_length": 299.125,
|
|
"eval_completions/mean_terminated_length": 287.9608660888672,
|
|
"eval_completions/min_length": 189.2,
|
|
"eval_completions/min_terminated_length": 189.2,
|
|
"eval_frac_reward_zero_std": 0.52,
|
|
"eval_kl": 1.4156181728839874,
|
|
"eval_loss": 0.006139134522527456,
|
|
"eval_num_tokens": 8205108.0,
|
|
"eval_reward": 3.146999979019165,
|
|
"eval_reward_std": 0.6526723924279213,
|
|
"eval_rewards/correctness_reward_func/mean": 1.6575,
|
|
"eval_rewards/correctness_reward_func/std": 1.4011952543258668,
|
|
"eval_rewards/format_reward_func/mean": 0.5,
|
|
"eval_rewards/format_reward_func/std": 0.0,
|
|
"eval_rewards/int_reward_func/mean": 0.5,
|
|
"eval_rewards/int_reward_func/std": 0.0,
|
|
"eval_rewards/think_structure_reward_func/mean": 0.48950000166893004,
|
|
"eval_rewards/think_structure_reward_func/std": 0.028225075155496597,
|
|
"eval_runtime": 368.1275,
|
|
"eval_samples_per_second": 0.272,
|
|
"eval_steps_per_second": 0.019,
|
|
"step": 170
|
|
},
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.047222222222222235,
|
|
"completions/max_length": 501.6,
|
|
"completions/max_terminated_length": 482.2,
|
|
"completions/mean_length": 285.6222229003906,
|
|
"completions/mean_terminated_length": 274.36187744140625,
|
|
"completions/min_length": 138.4,
|
|
"completions/min_terminated_length": 138.4,
|
|
"epoch": 0.42717656631407647,
|
|
"frac_reward_zero_std": 0.6555555582046508,
|
|
"grad_norm": 1.1139501333236694,
|
|
"kl": 0.877180474003156,
|
|
"learning_rate": 7.12273167039238e-07,
|
|
"loss": 0.0212,
|
|
"num_tokens": 8420652.0,
|
|
"reward": 3.5572222232818604,
|
|
"reward_std": 0.43106929659843446,
|
|
"rewards/correctness_reward_func/mean": 2.0666666984558106,
|
|
"rewards/correctness_reward_func/std": 1.386830949783325,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.49055554866790774,
|
|
"rewards/think_structure_reward_func/std": 0.03229331895709038,
|
|
"step": 175
|
|
},
|
|
{
|
|
"clip_ratio": 0.0,
|
|
"completions/clipped_ratio": 0.01944444444444442,
|
|
"completions/max_length": 503.2,
|
|
"completions/max_terminated_length": 481.8,
|
|
"completions/mean_length": 282.9805633544922,
|
|
"completions/mean_terminated_length": 278.5406219482422,
|
|
"completions/min_length": 136.8,
|
|
"completions/min_terminated_length": 136.8,
|
|
"epoch": 0.43938161106590723,
|
|
"frac_reward_zero_std": 0.6777777910232544,
|
|
"grad_norm": 0.8358107209205627,
|
|
"kl": 0.6044944907228152,
|
|
"learning_rate": 6.928155622440679e-07,
|
|
"loss": 0.0085,
|
|
"num_tokens": 8635621.0,
|
|
"reward": 3.5627777576446533,
|
|
"reward_std": 0.422798915207386,
|
|
"rewards/correctness_reward_func/mean": 2.066666674613953,
|
|
"rewards/correctness_reward_func/std": 1.3673566102981567,
|
|
"rewards/format_reward_func/mean": 0.5,
|
|
"rewards/format_reward_func/std": 0.0,
|
|
"rewards/int_reward_func/mean": 0.5,
|
|
"rewards/int_reward_func/std": 0.0,
|
|
"rewards/think_structure_reward_func/mean": 0.4961111068725586,
|
|
"rewards/think_structure_reward_func/std": 0.024096784740686418,
|
|
"step": 180
|
|
},
|
|
{
|
|
"epoch": 0.43938161106590723,
|
|
"eval_clip_ratio": 0.0,
|
|
"eval_completions/clipped_ratio": 0.045,
|
|
"eval_completions/max_length": 445.36,
|
|
"eval_completions/max_terminated_length": 429.44,
|
|
"eval_completions/mean_length": 296.335,
|
|
"eval_completions/mean_terminated_length": 286.33893005371095,
|
|
"eval_completions/min_length": 178.44,
|
|
"eval_completions/min_terminated_length": 178.44,
|
|
"eval_frac_reward_zero_std": 0.59,
|
|
"eval_kl": 1.7828838777542115,
|
|
"eval_loss": 0.011470726691186428,
|
|
"eval_num_tokens": 8635621.0,
|
|
"eval_reward": 3.2254999923706054,
|
|
"eval_reward_std": 0.5682479813694954,
|
|
"eval_rewards/correctness_reward_func/mean": 1.7325,
|
|
"eval_rewards/correctness_reward_func/std": 1.377410478591919,
|
|
"eval_rewards/format_reward_func/mean": 0.5,
|
|
"eval_rewards/format_reward_func/std": 0.0,
|
|
"eval_rewards/int_reward_func/mean": 0.5,
|
|
"eval_rewards/int_reward_func/std": 0.0,
|
|
"eval_rewards/think_structure_reward_func/mean": 0.4930000019073486,
|
|
"eval_rewards/think_structure_reward_func/std": 0.019914846420288086,
|
|
"eval_runtime": 375.0337,
|
|
"eval_samples_per_second": 0.267,
|
|
"eval_steps_per_second": 0.019,
|
|
"step": 180
|
|
}
|
|
],
|
|
"logging_steps": 5,
|
|
"max_steps": 410,
|
|
"num_input_tokens_seen": 8635621,
|
|
"num_train_epochs": 1,
|
|
"save_steps": 20,
|
|
"stateful_callbacks": {
|
|
"TrainerControl": {
|
|
"args": {
|
|
"should_epoch_stop": false,
|
|
"should_evaluate": false,
|
|
"should_log": false,
|
|
"should_save": true,
|
|
"should_training_stop": false
|
|
},
|
|
"attributes": {}
|
|
}
|
|
},
|
|
"total_flos": 0.0,
|
|
"train_batch_size": 6,
|
|
"trial_name": null,
|
|
"trial_params": null
|
|
}
|