Model: hophop1/Qwen2.5-0.5B-Instruct-Gensyn-Swarm-winged_fanged_mallard Source: Original Platform
824 lines
34 KiB
JSON
824 lines
34 KiB
JSON
{
|
|
"best_global_step": null,
|
|
"best_metric": null,
|
|
"best_model_checkpoint": null,
|
|
"epoch": 19.666666666666668,
|
|
"eval_steps": 500,
|
|
"global_step": 20,
|
|
"is_hyper_param_search": false,
|
|
"is_local_process_zero": true,
|
|
"is_world_process_zero": true,
|
|
"log_history": [
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0,
|
|
"completions/max_length": 32.0,
|
|
"completions/max_terminated_length": 32.0,
|
|
"completions/mean_length": 28.33333396911621,
|
|
"completions/mean_terminated_length": 28.33333396911621,
|
|
"completions/min_length": 25.0,
|
|
"completions/min_terminated_length": 25.0,
|
|
"epoch": 0.6666666666666666,
|
|
"grad_norm": 23.18532371520996,
|
|
"kl": 0.0,
|
|
"learning_rate": 0.0,
|
|
"loss": -0.0239,
|
|
"num_tokens": 628.0,
|
|
"reward": 1.2813235521316528,
|
|
"reward_std": 0.008508604019880295,
|
|
"rewards/concensus_correctness_reward_func/mean": 0.0,
|
|
"rewards/concensus_correctness_reward_func/std": 0.0,
|
|
"rewards/consensus_reward_func/mean": 0.0,
|
|
"rewards/consensus_reward_func/std": 0.0,
|
|
"rewards/cumulative_reward_2/mean": 0.0,
|
|
"rewards/cumulative_reward_2/std": 0.0,
|
|
"rewards/final_correctness_reward_func/mean": 1.0,
|
|
"rewards/final_correctness_reward_func/std": 1.154700517654419,
|
|
"rewards/question_recreation_reward_func/mean": 0.03132347762584686,
|
|
"rewards/question_recreation_reward_func/std": 0.008864727802574635,
|
|
"rewards/soft_format_reward_func/mean": 0.0,
|
|
"rewards/soft_format_reward_func/std": 0.0,
|
|
"rewards/strict_format_reward_func/mean": 0.0,
|
|
"rewards/strict_format_reward_func/std": 0.0,
|
|
"rewards/xmlcount_reward_func/mean": 0.25,
|
|
"rewards/xmlcount_reward_func/std": 0.28867512941360474,
|
|
"step": 1
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0,
|
|
"completions/max_length": 29.0,
|
|
"completions/max_terminated_length": 29.0,
|
|
"completions/mean_length": 23.0,
|
|
"completions/mean_terminated_length": 23.0,
|
|
"completions/min_length": 15.0,
|
|
"completions/min_terminated_length": 15.0,
|
|
"epoch": 1.6666666666666665,
|
|
"grad_norm": 28.710390090942383,
|
|
"kl": 0.0,
|
|
"learning_rate": 1e-06,
|
|
"loss": 0.1115,
|
|
"num_tokens": 1234.0,
|
|
"reward": 1.284814715385437,
|
|
"reward_std": 0.005810028873383999,
|
|
"rewards/concensus_correctness_reward_func/mean": 0.0,
|
|
"rewards/concensus_correctness_reward_func/std": 0.0,
|
|
"rewards/consensus_reward_func/mean": 0.0,
|
|
"rewards/consensus_reward_func/std": 0.0,
|
|
"rewards/cumulative_reward_2/mean": 0.0,
|
|
"rewards/cumulative_reward_2/std": 0.0,
|
|
"rewards/final_correctness_reward_func/mean": 1.0,
|
|
"rewards/final_correctness_reward_func/std": 1.154700517654419,
|
|
"rewards/question_recreation_reward_func/mean": 0.034814756363630295,
|
|
"rewards/question_recreation_reward_func/std": 0.024560028687119484,
|
|
"rewards/soft_format_reward_func/mean": 0.0,
|
|
"rewards/soft_format_reward_func/std": 0.0,
|
|
"rewards/strict_format_reward_func/mean": 0.0,
|
|
"rewards/strict_format_reward_func/std": 0.0,
|
|
"rewards/xmlcount_reward_func/mean": 0.25,
|
|
"rewards/xmlcount_reward_func/std": 0.28867512941360474,
|
|
"step": 2
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0,
|
|
"completions/max_length": 37.0,
|
|
"completions/max_terminated_length": 37.0,
|
|
"completions/mean_length": 33.66666793823242,
|
|
"completions/mean_terminated_length": 33.66666793823242,
|
|
"completions/min_length": 32.0,
|
|
"completions/min_terminated_length": 32.0,
|
|
"epoch": 2.6666666666666665,
|
|
"grad_norm": 10.300809860229492,
|
|
"kl": 0.010776982642710209,
|
|
"learning_rate": 9.931806517013612e-07,
|
|
"loss": 0.0224,
|
|
"num_tokens": 1889.0,
|
|
"reward": 1.2809349298477173,
|
|
"reward_std": 0.002600151114165783,
|
|
"rewards/concensus_correctness_reward_func/mean": 0.0,
|
|
"rewards/concensus_correctness_reward_func/std": 0.0,
|
|
"rewards/consensus_reward_func/mean": 0.0,
|
|
"rewards/consensus_reward_func/std": 0.0,
|
|
"rewards/cumulative_reward_2/mean": 0.0,
|
|
"rewards/cumulative_reward_2/std": 0.0,
|
|
"rewards/final_correctness_reward_func/mean": 1.0,
|
|
"rewards/final_correctness_reward_func/std": 1.154700517654419,
|
|
"rewards/question_recreation_reward_func/mean": 0.030935000628232956,
|
|
"rewards/question_recreation_reward_func/std": 0.005902186036109924,
|
|
"rewards/soft_format_reward_func/mean": 0.0,
|
|
"rewards/soft_format_reward_func/std": 0.0,
|
|
"rewards/strict_format_reward_func/mean": 0.0,
|
|
"rewards/strict_format_reward_func/std": 0.0,
|
|
"rewards/xmlcount_reward_func/mean": 0.25,
|
|
"rewards/xmlcount_reward_func/std": 0.28867512941360474,
|
|
"step": 3
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0,
|
|
"completions/max_length": 32.0,
|
|
"completions/max_terminated_length": 32.0,
|
|
"completions/mean_length": 18.33333396911621,
|
|
"completions/mean_terminated_length": 18.33333396911621,
|
|
"completions/min_length": 10.0,
|
|
"completions/min_terminated_length": 10.0,
|
|
"epoch": 3.6666666666666665,
|
|
"grad_norm": 34.62272262573242,
|
|
"kl": 0.16381593234837055,
|
|
"learning_rate": 9.729086208503173e-07,
|
|
"loss": -0.0388,
|
|
"num_tokens": 2488.0,
|
|
"reward": 1.2828543186187744,
|
|
"reward_std": 0.002999177435413003,
|
|
"rewards/concensus_correctness_reward_func/mean": 0.0,
|
|
"rewards/concensus_correctness_reward_func/std": 0.0,
|
|
"rewards/consensus_reward_func/mean": 0.0,
|
|
"rewards/consensus_reward_func/std": 0.0,
|
|
"rewards/cumulative_reward_2/mean": 0.0,
|
|
"rewards/cumulative_reward_2/std": 0.0,
|
|
"rewards/final_correctness_reward_func/mean": 1.0,
|
|
"rewards/final_correctness_reward_func/std": 1.154700517654419,
|
|
"rewards/question_recreation_reward_func/mean": 0.032854292541742325,
|
|
"rewards/question_recreation_reward_func/std": 0.01772441901266575,
|
|
"rewards/soft_format_reward_func/mean": 0.0,
|
|
"rewards/soft_format_reward_func/std": 0.0,
|
|
"rewards/strict_format_reward_func/mean": 0.0,
|
|
"rewards/strict_format_reward_func/std": 0.0,
|
|
"rewards/xmlcount_reward_func/mean": 0.25,
|
|
"rewards/xmlcount_reward_func/std": 0.28867512941360474,
|
|
"step": 4
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0,
|
|
"completions/max_length": 33.0,
|
|
"completions/max_terminated_length": 33.0,
|
|
"completions/mean_length": 25.0,
|
|
"completions/mean_terminated_length": 25.0,
|
|
"completions/min_length": 14.0,
|
|
"completions/min_terminated_length": 14.0,
|
|
"epoch": 4.666666666666667,
|
|
"grad_norm": 28.476211547851562,
|
|
"kl": 0.10297379642724991,
|
|
"learning_rate": 9.397368756032444e-07,
|
|
"loss": 0.1969,
|
|
"num_tokens": 3127.0,
|
|
"reward": 0.04856175556778908,
|
|
"reward_std": 0.0062292092479765415,
|
|
"rewards/concensus_correctness_reward_func/mean": 0.0,
|
|
"rewards/concensus_correctness_reward_func/std": 0.0,
|
|
"rewards/consensus_reward_func/mean": 0.0,
|
|
"rewards/consensus_reward_func/std": 0.0,
|
|
"rewards/cumulative_reward_2/mean": 0.0,
|
|
"rewards/cumulative_reward_2/std": 0.0,
|
|
"rewards/final_correctness_reward_func/mean": 0.0,
|
|
"rewards/final_correctness_reward_func/std": 0.0,
|
|
"rewards/question_recreation_reward_func/mean": 0.04856175556778908,
|
|
"rewards/question_recreation_reward_func/std": 0.011551093310117722,
|
|
"rewards/soft_format_reward_func/mean": 0.0,
|
|
"rewards/soft_format_reward_func/std": 0.0,
|
|
"rewards/strict_format_reward_func/mean": 0.0,
|
|
"rewards/strict_format_reward_func/std": 0.0,
|
|
"rewards/xmlcount_reward_func/mean": 0.0,
|
|
"rewards/xmlcount_reward_func/std": 0.0,
|
|
"step": 5
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0,
|
|
"completions/max_length": 32.0,
|
|
"completions/max_terminated_length": 32.0,
|
|
"completions/mean_length": 24.33333396911621,
|
|
"completions/mean_terminated_length": 24.33333396911621,
|
|
"completions/min_length": 16.0,
|
|
"completions/min_terminated_length": 16.0,
|
|
"epoch": 5.666666666666667,
|
|
"grad_norm": 29.99322509765625,
|
|
"kl": 0.06120647024363279,
|
|
"learning_rate": 8.945702546981968e-07,
|
|
"loss": -0.0746,
|
|
"num_tokens": 3744.0,
|
|
"reward": 1.2867157459259033,
|
|
"reward_std": 0.007329343352466822,
|
|
"rewards/concensus_correctness_reward_func/mean": 0.0,
|
|
"rewards/concensus_correctness_reward_func/std": 0.0,
|
|
"rewards/consensus_reward_func/mean": 0.0,
|
|
"rewards/consensus_reward_func/std": 0.0,
|
|
"rewards/cumulative_reward_2/mean": 0.0,
|
|
"rewards/cumulative_reward_2/std": 0.0,
|
|
"rewards/final_correctness_reward_func/mean": 1.0,
|
|
"rewards/final_correctness_reward_func/std": 1.154700517654419,
|
|
"rewards/question_recreation_reward_func/mean": 0.03671575337648392,
|
|
"rewards/question_recreation_reward_func/std": 0.023405639454722404,
|
|
"rewards/soft_format_reward_func/mean": 0.0,
|
|
"rewards/soft_format_reward_func/std": 0.0,
|
|
"rewards/strict_format_reward_func/mean": 0.0,
|
|
"rewards/strict_format_reward_func/std": 0.0,
|
|
"rewards/xmlcount_reward_func/mean": 0.25,
|
|
"rewards/xmlcount_reward_func/std": 0.28867512941360474,
|
|
"step": 6
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0,
|
|
"completions/max_length": 49.0,
|
|
"completions/max_terminated_length": 49.0,
|
|
"completions/mean_length": 24.0,
|
|
"completions/mean_terminated_length": 24.0,
|
|
"completions/min_length": 9.0,
|
|
"completions/min_terminated_length": 9.0,
|
|
"epoch": 6.666666666666667,
|
|
"grad_norm": 28.319862365722656,
|
|
"kl": 0.1611136607825756,
|
|
"learning_rate": 8.386407858128706e-07,
|
|
"loss": -0.2244,
|
|
"num_tokens": 4343.0,
|
|
"reward": 0.026412304490804672,
|
|
"reward_std": 0.013399245217442513,
|
|
"rewards/concensus_correctness_reward_func/mean": 0.0,
|
|
"rewards/concensus_correctness_reward_func/std": 0.0,
|
|
"rewards/consensus_reward_func/mean": 0.0,
|
|
"rewards/consensus_reward_func/std": 0.0,
|
|
"rewards/cumulative_reward_2/mean": 0.0,
|
|
"rewards/cumulative_reward_2/std": 0.0,
|
|
"rewards/final_correctness_reward_func/mean": 0.0,
|
|
"rewards/final_correctness_reward_func/std": 0.0,
|
|
"rewards/question_recreation_reward_func/mean": 0.026412304490804672,
|
|
"rewards/question_recreation_reward_func/std": 0.013973159715533257,
|
|
"rewards/soft_format_reward_func/mean": 0.0,
|
|
"rewards/soft_format_reward_func/std": 0.0,
|
|
"rewards/strict_format_reward_func/mean": 0.0,
|
|
"rewards/strict_format_reward_func/std": 0.0,
|
|
"rewards/xmlcount_reward_func/mean": 0.0,
|
|
"rewards/xmlcount_reward_func/std": 0.0,
|
|
"step": 7
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0,
|
|
"completions/max_length": 32.0,
|
|
"completions/max_terminated_length": 32.0,
|
|
"completions/mean_length": 29.0,
|
|
"completions/mean_terminated_length": 29.0,
|
|
"completions/min_length": 23.0,
|
|
"completions/min_terminated_length": 23.0,
|
|
"epoch": 7.666666666666667,
|
|
"grad_norm": 20.72979164123535,
|
|
"kl": 0.054506564512848854,
|
|
"learning_rate": 7.734740790612136e-07,
|
|
"loss": -0.0114,
|
|
"num_tokens": 4967.0,
|
|
"reward": 1.2918072938919067,
|
|
"reward_std": 0.0005663956981152296,
|
|
"rewards/concensus_correctness_reward_func/mean": 0.0,
|
|
"rewards/concensus_correctness_reward_func/std": 0.0,
|
|
"rewards/consensus_reward_func/mean": 0.0,
|
|
"rewards/consensus_reward_func/std": 0.0,
|
|
"rewards/cumulative_reward_2/mean": 0.0,
|
|
"rewards/cumulative_reward_2/std": 0.0,
|
|
"rewards/final_correctness_reward_func/mean": 1.0,
|
|
"rewards/final_correctness_reward_func/std": 1.154700517654419,
|
|
"rewards/question_recreation_reward_func/mean": 0.04180721938610077,
|
|
"rewards/question_recreation_reward_func/std": 0.00750124454498291,
|
|
"rewards/soft_format_reward_func/mean": 0.0,
|
|
"rewards/soft_format_reward_func/std": 0.0,
|
|
"rewards/strict_format_reward_func/mean": 0.0,
|
|
"rewards/strict_format_reward_func/std": 0.0,
|
|
"rewards/xmlcount_reward_func/mean": 0.25,
|
|
"rewards/xmlcount_reward_func/std": 0.28867512941360474,
|
|
"step": 8
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0,
|
|
"completions/max_length": 32.0,
|
|
"completions/max_terminated_length": 32.0,
|
|
"completions/mean_length": 16.66666603088379,
|
|
"completions/mean_terminated_length": 16.66666603088379,
|
|
"completions/min_length": 9.0,
|
|
"completions/min_terminated_length": 9.0,
|
|
"epoch": 8.666666666666666,
|
|
"grad_norm": 16.976764678955078,
|
|
"kl": 0.4037783732637763,
|
|
"learning_rate": 7.008477123264847e-07,
|
|
"loss": 0.0162,
|
|
"num_tokens": 5561.0,
|
|
"reward": 1.832707405090332,
|
|
"reward_std": 0.7097353935241699,
|
|
"rewards/concensus_correctness_reward_func/mean": 0.0,
|
|
"rewards/concensus_correctness_reward_func/std": 0.0,
|
|
"rewards/consensus_reward_func/mean": 0.0,
|
|
"rewards/consensus_reward_func/std": 0.0,
|
|
"rewards/cumulative_reward_2/mean": 0.0,
|
|
"rewards/cumulative_reward_2/std": 0.0,
|
|
"rewards/final_correctness_reward_func/mean": 1.5,
|
|
"rewards/final_correctness_reward_func/std": 1.0,
|
|
"rewards/question_recreation_reward_func/mean": 0.02020736038684845,
|
|
"rewards/question_recreation_reward_func/std": 0.01928129233419895,
|
|
"rewards/soft_format_reward_func/mean": 0.0,
|
|
"rewards/soft_format_reward_func/std": 0.0,
|
|
"rewards/strict_format_reward_func/mean": 0.0,
|
|
"rewards/strict_format_reward_func/std": 0.0,
|
|
"rewards/xmlcount_reward_func/mean": 0.3125,
|
|
"rewards/xmlcount_reward_func/std": 0.21650634706020355,
|
|
"step": 9
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0,
|
|
"completions/max_length": 32.0,
|
|
"completions/max_terminated_length": 32.0,
|
|
"completions/mean_length": 29.33333396911621,
|
|
"completions/mean_terminated_length": 29.33333396911621,
|
|
"completions/min_length": 24.0,
|
|
"completions/min_terminated_length": 24.0,
|
|
"epoch": 9.666666666666666,
|
|
"grad_norm": 27.643352508544922,
|
|
"kl": 0.14844292961061,
|
|
"learning_rate": 6.227427435703995e-07,
|
|
"loss": 0.0466,
|
|
"num_tokens": 6180.0,
|
|
"reward": 1.301353096961975,
|
|
"reward_std": 0.004389901179820299,
|
|
"rewards/concensus_correctness_reward_func/mean": 0.0,
|
|
"rewards/concensus_correctness_reward_func/std": 0.0,
|
|
"rewards/consensus_reward_func/mean": 0.0,
|
|
"rewards/consensus_reward_func/std": 0.0,
|
|
"rewards/cumulative_reward_2/mean": 0.0,
|
|
"rewards/cumulative_reward_2/std": 0.0,
|
|
"rewards/final_correctness_reward_func/mean": 1.0,
|
|
"rewards/final_correctness_reward_func/std": 1.154700517654419,
|
|
"rewards/question_recreation_reward_func/mean": 0.05135301500558853,
|
|
"rewards/question_recreation_reward_func/std": 0.019177274778485298,
|
|
"rewards/soft_format_reward_func/mean": 0.0,
|
|
"rewards/soft_format_reward_func/std": 0.0,
|
|
"rewards/strict_format_reward_func/mean": 0.0,
|
|
"rewards/strict_format_reward_func/std": 0.0,
|
|
"rewards/xmlcount_reward_func/mean": 0.25,
|
|
"rewards/xmlcount_reward_func/std": 0.28867512941360474,
|
|
"step": 10
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0,
|
|
"completions/max_length": 32.0,
|
|
"completions/max_terminated_length": 32.0,
|
|
"completions/mean_length": 27.0,
|
|
"completions/mean_terminated_length": 27.0,
|
|
"completions/min_length": 22.0,
|
|
"completions/min_terminated_length": 22.0,
|
|
"epoch": 10.666666666666666,
|
|
"grad_norm": 23.045936584472656,
|
|
"kl": 0.12309377361088991,
|
|
"learning_rate": 5.412896727361662e-07,
|
|
"loss": -0.0306,
|
|
"num_tokens": 6805.0,
|
|
"reward": 1.2927604913711548,
|
|
"reward_std": 0.0032423806842416525,
|
|
"rewards/concensus_correctness_reward_func/mean": 0.0,
|
|
"rewards/concensus_correctness_reward_func/std": 0.0,
|
|
"rewards/consensus_reward_func/mean": 0.0,
|
|
"rewards/consensus_reward_func/std": 0.0,
|
|
"rewards/cumulative_reward_2/mean": 0.0,
|
|
"rewards/cumulative_reward_2/std": 0.0,
|
|
"rewards/final_correctness_reward_func/mean": 1.0,
|
|
"rewards/final_correctness_reward_func/std": 1.154700517654419,
|
|
"rewards/question_recreation_reward_func/mean": 0.04276055097579956,
|
|
"rewards/question_recreation_reward_func/std": 0.02906481921672821,
|
|
"rewards/soft_format_reward_func/mean": 0.0,
|
|
"rewards/soft_format_reward_func/std": 0.0,
|
|
"rewards/strict_format_reward_func/mean": 0.0,
|
|
"rewards/strict_format_reward_func/std": 0.0,
|
|
"rewards/xmlcount_reward_func/mean": 0.25,
|
|
"rewards/xmlcount_reward_func/std": 0.28867512941360474,
|
|
"step": 11
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0,
|
|
"completions/max_length": 34.0,
|
|
"completions/max_terminated_length": 34.0,
|
|
"completions/mean_length": 28.33333396911621,
|
|
"completions/mean_terminated_length": 28.33333396911621,
|
|
"completions/min_length": 17.0,
|
|
"completions/min_terminated_length": 17.0,
|
|
"epoch": 11.666666666666666,
|
|
"grad_norm": 24.045764923095703,
|
|
"kl": 0.30065467208623886,
|
|
"learning_rate": 4.5871032726383385e-07,
|
|
"loss": -0.1085,
|
|
"num_tokens": 7437.0,
|
|
"reward": 0.5449972152709961,
|
|
"reward_std": 0.7033753395080566,
|
|
"rewards/concensus_correctness_reward_func/mean": 0.0,
|
|
"rewards/concensus_correctness_reward_func/std": 0.0,
|
|
"rewards/consensus_reward_func/mean": 0.0,
|
|
"rewards/consensus_reward_func/std": 0.0,
|
|
"rewards/cumulative_reward_2/mean": 0.0,
|
|
"rewards/cumulative_reward_2/std": 0.0,
|
|
"rewards/final_correctness_reward_func/mean": 0.5,
|
|
"rewards/final_correctness_reward_func/std": 1.0,
|
|
"rewards/question_recreation_reward_func/mean": 0.044997282326221466,
|
|
"rewards/question_recreation_reward_func/std": 0.011617259122431278,
|
|
"rewards/soft_format_reward_func/mean": 0.0,
|
|
"rewards/soft_format_reward_func/std": 0.0,
|
|
"rewards/strict_format_reward_func/mean": 0.0,
|
|
"rewards/strict_format_reward_func/std": 0.0,
|
|
"rewards/xmlcount_reward_func/mean": 0.0,
|
|
"rewards/xmlcount_reward_func/std": 0.0,
|
|
"step": 12
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0,
|
|
"completions/max_length": 40.0,
|
|
"completions/max_terminated_length": 40.0,
|
|
"completions/mean_length": 36.66666793823242,
|
|
"completions/mean_terminated_length": 36.66666793823242,
|
|
"completions/min_length": 32.0,
|
|
"completions/min_terminated_length": 32.0,
|
|
"epoch": 12.666666666666666,
|
|
"grad_norm": 7.944507598876953,
|
|
"kl": 0.1541438391432166,
|
|
"learning_rate": 3.772572564296004e-07,
|
|
"loss": 0.0149,
|
|
"num_tokens": 8091.0,
|
|
"reward": 1.7904198169708252,
|
|
"reward_std": 0.7157720923423767,
|
|
"rewards/concensus_correctness_reward_func/mean": 0.0,
|
|
"rewards/concensus_correctness_reward_func/std": 0.0,
|
|
"rewards/consensus_reward_func/mean": 0.0,
|
|
"rewards/consensus_reward_func/std": 0.0,
|
|
"rewards/cumulative_reward_2/mean": 0.0,
|
|
"rewards/cumulative_reward_2/std": 0.0,
|
|
"rewards/final_correctness_reward_func/mean": 1.5,
|
|
"rewards/final_correctness_reward_func/std": 1.0,
|
|
"rewards/question_recreation_reward_func/mean": 0.03316996619105339,
|
|
"rewards/question_recreation_reward_func/std": 0.0044271741062402725,
|
|
"rewards/soft_format_reward_func/mean": 0.0,
|
|
"rewards/soft_format_reward_func/std": 0.0,
|
|
"rewards/strict_format_reward_func/mean": 0.0,
|
|
"rewards/strict_format_reward_func/std": 0.0,
|
|
"rewards/xmlcount_reward_func/mean": 0.25725001096725464,
|
|
"rewards/xmlcount_reward_func/std": 0.2805534899234772,
|
|
"step": 13
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0,
|
|
"completions/max_length": 335.0,
|
|
"completions/max_terminated_length": 335.0,
|
|
"completions/mean_length": 135.6666717529297,
|
|
"completions/mean_terminated_length": 135.6666717529297,
|
|
"completions/min_length": 32.0,
|
|
"completions/min_terminated_length": 32.0,
|
|
"epoch": 13.666666666666666,
|
|
"grad_norm": 7.2156081199646,
|
|
"kl": 0.07004917785525322,
|
|
"learning_rate": 2.9915228767351535e-07,
|
|
"loss": 0.2809,
|
|
"num_tokens": 9042.0,
|
|
"reward": 1.7817018032073975,
|
|
"reward_std": 0.7197564244270325,
|
|
"rewards/concensus_correctness_reward_func/mean": 0.0,
|
|
"rewards/concensus_correctness_reward_func/std": 0.0,
|
|
"rewards/consensus_reward_func/mean": 0.0,
|
|
"rewards/consensus_reward_func/std": 0.0,
|
|
"rewards/cumulative_reward_2/mean": 0.0,
|
|
"rewards/cumulative_reward_2/std": 0.0,
|
|
"rewards/final_correctness_reward_func/mean": 1.5,
|
|
"rewards/final_correctness_reward_func/std": 1.0,
|
|
"rewards/question_recreation_reward_func/mean": 0.02720182202756405,
|
|
"rewards/question_recreation_reward_func/std": 0.013149062171578407,
|
|
"rewards/soft_format_reward_func/mean": 0.0,
|
|
"rewards/soft_format_reward_func/std": 0.0,
|
|
"rewards/strict_format_reward_func/mean": 0.0,
|
|
"rewards/strict_format_reward_func/std": 0.0,
|
|
"rewards/xmlcount_reward_func/mean": 0.25450000166893005,
|
|
"rewards/xmlcount_reward_func/std": 0.2835742235183716,
|
|
"step": 14
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0,
|
|
"completions/max_length": 46.0,
|
|
"completions/max_terminated_length": 46.0,
|
|
"completions/mean_length": 24.0,
|
|
"completions/mean_terminated_length": 24.0,
|
|
"completions/min_length": 9.0,
|
|
"completions/min_terminated_length": 9.0,
|
|
"epoch": 14.666666666666666,
|
|
"grad_norm": 39.25900650024414,
|
|
"kl": 0.3023638464510441,
|
|
"learning_rate": 2.2652592093878665e-07,
|
|
"loss": 0.1531,
|
|
"num_tokens": 9664.0,
|
|
"reward": 0.07192623615264893,
|
|
"reward_std": 0.028722776100039482,
|
|
"rewards/concensus_correctness_reward_func/mean": 0.0,
|
|
"rewards/concensus_correctness_reward_func/std": 0.0,
|
|
"rewards/consensus_reward_func/mean": 0.0,
|
|
"rewards/consensus_reward_func/std": 0.0,
|
|
"rewards/cumulative_reward_2/mean": 0.0,
|
|
"rewards/cumulative_reward_2/std": 0.0,
|
|
"rewards/final_correctness_reward_func/mean": 0.0,
|
|
"rewards/final_correctness_reward_func/std": 0.0,
|
|
"rewards/question_recreation_reward_func/mean": 0.04067623242735863,
|
|
"rewards/question_recreation_reward_func/std": 0.020941482856869698,
|
|
"rewards/soft_format_reward_func/mean": 0.0,
|
|
"rewards/soft_format_reward_func/std": 0.0,
|
|
"rewards/strict_format_reward_func/mean": 0.0,
|
|
"rewards/strict_format_reward_func/std": 0.0,
|
|
"rewards/xmlcount_reward_func/mean": 0.03125,
|
|
"rewards/xmlcount_reward_func/std": 0.0625,
|
|
"step": 15
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0,
|
|
"completions/max_length": 34.0,
|
|
"completions/max_terminated_length": 34.0,
|
|
"completions/mean_length": 32.66666793823242,
|
|
"completions/mean_terminated_length": 32.66666793823242,
|
|
"completions/min_length": 32.0,
|
|
"completions/min_terminated_length": 32.0,
|
|
"epoch": 15.666666666666666,
|
|
"grad_norm": 24.114459991455078,
|
|
"kl": 0.2044437825679779,
|
|
"learning_rate": 1.6135921418712955e-07,
|
|
"loss": -0.0388,
|
|
"num_tokens": 10300.0,
|
|
"reward": 1.2992045879364014,
|
|
"reward_std": 0.014325941912829876,
|
|
"rewards/concensus_correctness_reward_func/mean": 0.0,
|
|
"rewards/concensus_correctness_reward_func/std": 0.0,
|
|
"rewards/consensus_reward_func/mean": 0.0,
|
|
"rewards/consensus_reward_func/std": 0.0,
|
|
"rewards/cumulative_reward_2/mean": 0.0,
|
|
"rewards/cumulative_reward_2/std": 0.0,
|
|
"rewards/final_correctness_reward_func/mean": 1.0,
|
|
"rewards/final_correctness_reward_func/std": 1.154700517654419,
|
|
"rewards/question_recreation_reward_func/mean": 0.049204543232917786,
|
|
"rewards/question_recreation_reward_func/std": 0.023023977875709534,
|
|
"rewards/soft_format_reward_func/mean": 0.0,
|
|
"rewards/soft_format_reward_func/std": 0.0,
|
|
"rewards/strict_format_reward_func/mean": 0.0,
|
|
"rewards/strict_format_reward_func/std": 0.0,
|
|
"rewards/xmlcount_reward_func/mean": 0.25,
|
|
"rewards/xmlcount_reward_func/std": 0.28867512941360474,
|
|
"step": 16
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0,
|
|
"completions/max_length": 44.0,
|
|
"completions/max_terminated_length": 44.0,
|
|
"completions/mean_length": 32.66666793823242,
|
|
"completions/mean_terminated_length": 32.66666793823242,
|
|
"completions/min_length": 24.0,
|
|
"completions/min_terminated_length": 24.0,
|
|
"epoch": 16.666666666666668,
|
|
"grad_norm": 24.598146438598633,
|
|
"kl": 0.348018154501915,
|
|
"learning_rate": 1.0542974530180327e-07,
|
|
"loss": 0.0332,
|
|
"num_tokens": 10942.0,
|
|
"reward": 1.0598185062408447,
|
|
"reward_std": 0.008537341840565205,
|
|
"rewards/concensus_correctness_reward_func/mean": 0.0,
|
|
"rewards/concensus_correctness_reward_func/std": 0.0,
|
|
"rewards/consensus_reward_func/mean": 0.0,
|
|
"rewards/consensus_reward_func/std": 0.0,
|
|
"rewards/cumulative_reward_2/mean": 0.0,
|
|
"rewards/cumulative_reward_2/std": 0.0,
|
|
"rewards/final_correctness_reward_func/mean": 1.0,
|
|
"rewards/final_correctness_reward_func/std": 1.154700517654419,
|
|
"rewards/question_recreation_reward_func/mean": 0.04881836846470833,
|
|
"rewards/question_recreation_reward_func/std": 0.01848919875919819,
|
|
"rewards/soft_format_reward_func/mean": 0.0,
|
|
"rewards/soft_format_reward_func/std": 0.0,
|
|
"rewards/strict_format_reward_func/mean": 0.0,
|
|
"rewards/strict_format_reward_func/std": 0.0,
|
|
"rewards/xmlcount_reward_func/mean": 0.010999999940395355,
|
|
"rewards/xmlcount_reward_func/std": 0.020688161253929138,
|
|
"step": 17
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0,
|
|
"completions/max_length": 45.0,
|
|
"completions/max_terminated_length": 45.0,
|
|
"completions/mean_length": 42.0,
|
|
"completions/mean_terminated_length": 42.0,
|
|
"completions/min_length": 40.0,
|
|
"completions/min_terminated_length": 40.0,
|
|
"epoch": 17.666666666666668,
|
|
"grad_norm": 20.540817260742188,
|
|
"kl": 0.3694180101156235,
|
|
"learning_rate": 6.026312439675551e-08,
|
|
"loss": -0.0307,
|
|
"num_tokens": 11638.0,
|
|
"reward": 1.0319881439208984,
|
|
"reward_std": 0.002910168608650565,
|
|
"rewards/concensus_correctness_reward_func/mean": 0.0,
|
|
"rewards/concensus_correctness_reward_func/std": 0.0,
|
|
"rewards/consensus_reward_func/mean": 0.0,
|
|
"rewards/consensus_reward_func/std": 0.0,
|
|
"rewards/cumulative_reward_2/mean": 0.0,
|
|
"rewards/cumulative_reward_2/std": 0.0,
|
|
"rewards/final_correctness_reward_func/mean": 1.0,
|
|
"rewards/final_correctness_reward_func/std": 1.154700517654419,
|
|
"rewards/question_recreation_reward_func/mean": 0.029738131910562515,
|
|
"rewards/question_recreation_reward_func/std": 0.0061986916698515415,
|
|
"rewards/soft_format_reward_func/mean": 0.0,
|
|
"rewards/soft_format_reward_func/std": 0.0,
|
|
"rewards/strict_format_reward_func/mean": 0.0,
|
|
"rewards/strict_format_reward_func/std": 0.0,
|
|
"rewards/xmlcount_reward_func/mean": 0.0022499999031424522,
|
|
"rewards/xmlcount_reward_func/std": 0.006652066949754953,
|
|
"step": 18
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0,
|
|
"completions/max_length": 41.0,
|
|
"completions/max_terminated_length": 41.0,
|
|
"completions/mean_length": 37.0,
|
|
"completions/mean_terminated_length": 37.0,
|
|
"completions/min_length": 30.0,
|
|
"completions/min_terminated_length": 30.0,
|
|
"epoch": 18.666666666666668,
|
|
"grad_norm": 27.271289825439453,
|
|
"kl": 0.34695659577846527,
|
|
"learning_rate": 2.7091379149682682e-08,
|
|
"loss": 0.0705,
|
|
"num_tokens": 12302.0,
|
|
"reward": 0.532687783241272,
|
|
"reward_std": 0.7111693024635315,
|
|
"rewards/concensus_correctness_reward_func/mean": 0.0,
|
|
"rewards/concensus_correctness_reward_func/std": 0.0,
|
|
"rewards/consensus_reward_func/mean": 0.0,
|
|
"rewards/consensus_reward_func/std": 0.0,
|
|
"rewards/cumulative_reward_2/mean": 0.0,
|
|
"rewards/cumulative_reward_2/std": 0.0,
|
|
"rewards/final_correctness_reward_func/mean": 0.5,
|
|
"rewards/final_correctness_reward_func/std": 1.0,
|
|
"rewards/question_recreation_reward_func/mean": 0.0281878300011158,
|
|
"rewards/question_recreation_reward_func/std": 0.005380373448133469,
|
|
"rewards/soft_format_reward_func/mean": 0.0,
|
|
"rewards/soft_format_reward_func/std": 0.0,
|
|
"rewards/strict_format_reward_func/mean": 0.0,
|
|
"rewards/strict_format_reward_func/std": 0.0,
|
|
"rewards/xmlcount_reward_func/mean": 0.0044999998062849045,
|
|
"rewards/xmlcount_reward_func/std": 0.008999999612569809,
|
|
"step": 19
|
|
},
|
|
{
|
|
"clip_ratio/high_max": 0.0,
|
|
"clip_ratio/high_mean": 0.0,
|
|
"clip_ratio/low_mean": 0.0,
|
|
"clip_ratio/low_min": 0.0,
|
|
"clip_ratio/region_mean": 0.0,
|
|
"completions/clipped_ratio": 0.0,
|
|
"completions/max_length": 40.0,
|
|
"completions/max_terminated_length": 40.0,
|
|
"completions/mean_length": 34.0,
|
|
"completions/mean_terminated_length": 34.0,
|
|
"completions/min_length": 30.0,
|
|
"completions/min_terminated_length": 30.0,
|
|
"epoch": 19.666666666666668,
|
|
"grad_norm": 20.589397430419922,
|
|
"kl": 0.16628471948206425,
|
|
"learning_rate": 6.819348298638839e-09,
|
|
"loss": 0.057,
|
|
"num_tokens": 12948.0,
|
|
"reward": 1.2811635732650757,
|
|
"reward_std": 0.014331253245472908,
|
|
"rewards/concensus_correctness_reward_func/mean": 0.0,
|
|
"rewards/concensus_correctness_reward_func/std": 0.0,
|
|
"rewards/consensus_reward_func/mean": 0.0,
|
|
"rewards/consensus_reward_func/std": 0.0,
|
|
"rewards/cumulative_reward_2/mean": 0.0,
|
|
"rewards/cumulative_reward_2/std": 0.0,
|
|
"rewards/final_correctness_reward_func/mean": 1.0,
|
|
"rewards/final_correctness_reward_func/std": 1.154700517654419,
|
|
"rewards/question_recreation_reward_func/mean": 0.031163623556494713,
|
|
"rewards/question_recreation_reward_func/std": 0.02261289581656456,
|
|
"rewards/soft_format_reward_func/mean": 0.0,
|
|
"rewards/soft_format_reward_func/std": 0.0,
|
|
"rewards/strict_format_reward_func/mean": 0.0,
|
|
"rewards/strict_format_reward_func/std": 0.0,
|
|
"rewards/xmlcount_reward_func/mean": 0.25,
|
|
"rewards/xmlcount_reward_func/std": 0.28867512941360474,
|
|
"step": 20
|
|
},
|
|
{
|
|
"epoch": 19.666666666666668,
|
|
"step": 20,
|
|
"total_flos": 0.0,
|
|
"train_loss": 0.02108212043531239,
|
|
"train_runtime": 15960.1034,
|
|
"train_samples_per_second": 0.005,
|
|
"train_steps_per_second": 0.001
|
|
}
|
|
],
|
|
"logging_steps": 1,
|
|
"max_steps": 20,
|
|
"num_input_tokens_seen": 12948,
|
|
"num_train_epochs": 20,
|
|
"save_steps": 25,
|
|
"stateful_callbacks": {
|
|
"TrainerControl": {
|
|
"args": {
|
|
"should_epoch_stop": false,
|
|
"should_evaluate": false,
|
|
"should_log": false,
|
|
"should_save": true,
|
|
"should_training_stop": true
|
|
},
|
|
"attributes": {}
|
|
}
|
|
},
|
|
"total_flos": 0.0,
|
|
"train_batch_size": 2,
|
|
"trial_name": null,
|
|
"trial_params": null
|
|
}
|