Files
Qwen2.5-0.5B-Instruct-Gensy…/trainer_state.json
ModelHub XC 9d14613e56 初始化项目,由ModelHub XC社区提供模型
Model: Naperzop/Qwen2.5-0.5B-Instruct-Gensyn-Swarm-shy_sprightly_robin
Source: Original Platform
2026-08-28 11:52:19 +08:00

244 lines
9.4 KiB
JSON

{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 0.8695652173913043,
"eval_steps": 500,
"global_step": 10,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 84.5,
"completions/max_terminated_length": 84.5,
"completions/mean_length": 46.125,
"completions/mean_terminated_length": 46.125,
"completions/min_length": 20.5,
"completions/min_terminated_length": 20.5,
"epoch": 0.17391304347826086,
"frac_reward_zero_std": 0.0,
"grad_norm": 50.021663665771484,
"kl": 0.0,
"learning_rate": 5e-07,
"loss": -0.0009,
"num_tokens": 1393.0,
"reward": 0.3569868355989456,
"reward_std": 0.10140731558203697,
"rewards/concensus_correctness_reward_func/mean": 0.0,
"rewards/concensus_correctness_reward_func/std": 0.0,
"rewards/consensus_reward_func/mean": 0.0,
"rewards/consensus_reward_func/std": 0.0,
"rewards/cumulative_reward_2/mean": 0.0,
"rewards/cumulative_reward_2/std": 0.0,
"rewards/final_correctness_reward_func/mean": 0.0,
"rewards/final_correctness_reward_func/std": 0.0,
"rewards/question_recreation_reward_func/mean": 0.032236830331385136,
"rewards/question_recreation_reward_func/std": 0.017948506865650415,
"rewards/soft_format_reward_func/mean": 0.0,
"rewards/soft_format_reward_func/std": 0.0,
"rewards/strict_format_reward_func/mean": 0.0,
"rewards/strict_format_reward_func/std": 0.0,
"rewards/xmlcount_reward_func/mean": 0.32474999129772186,
"rewards/xmlcount_reward_func/std": 0.19420991092920303,
"step": 2
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 109.0,
"completions/max_terminated_length": 109.0,
"completions/mean_length": 70.75,
"completions/mean_terminated_length": 70.75,
"completions/min_length": 41.5,
"completions/min_terminated_length": 41.5,
"epoch": 0.34782608695652173,
"frac_reward_zero_std": 0.0,
"grad_norm": 19.0537109375,
"kl": 0.002165338257327676,
"learning_rate": 4.415111107797445e-07,
"loss": 0.0282,
"num_tokens": 2983.0,
"reward": 0.4277445673942566,
"reward_std": 0.14516344666481018,
"rewards/concensus_correctness_reward_func/mean": 0.0,
"rewards/concensus_correctness_reward_func/std": 0.0,
"rewards/consensus_reward_func/mean": 0.0,
"rewards/consensus_reward_func/std": 0.0,
"rewards/cumulative_reward_2/mean": 0.0,
"rewards/cumulative_reward_2/std": 0.0,
"rewards/final_correctness_reward_func/mean": 0.0,
"rewards/final_correctness_reward_func/std": 0.0,
"rewards/question_recreation_reward_func/mean": 0.16299461387097836,
"rewards/question_recreation_reward_func/std": 0.06881882064044476,
"rewards/soft_format_reward_func/mean": 0.0,
"rewards/soft_format_reward_func/std": 0.0,
"rewards/strict_format_reward_func/mean": 0.0,
"rewards/strict_format_reward_func/std": 0.0,
"rewards/xmlcount_reward_func/mean": 0.26474998891353607,
"rewards/xmlcount_reward_func/std": 0.28970315866172314,
"step": 4
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 177.0,
"completions/max_terminated_length": 177.0,
"completions/mean_length": 91.625,
"completions/mean_terminated_length": 91.625,
"completions/min_length": 33.5,
"completions/min_terminated_length": 33.5,
"epoch": 0.5217391304347826,
"frac_reward_zero_std": 0.0,
"grad_norm": 13.790047645568848,
"kl": 0.0031155047909123823,
"learning_rate": 2.934120444167326e-07,
"loss": 0.1559,
"num_tokens": 4740.0,
"reward": 0.25488629564642906,
"reward_std": 0.1210553664714098,
"rewards/concensus_correctness_reward_func/mean": 0.0,
"rewards/concensus_correctness_reward_func/std": 0.0,
"rewards/consensus_reward_func/mean": 0.0,
"rewards/consensus_reward_func/std": 0.0,
"rewards/cumulative_reward_2/mean": 0.0,
"rewards/cumulative_reward_2/std": 0.0,
"rewards/final_correctness_reward_func/mean": 0.0,
"rewards/final_correctness_reward_func/std": 0.0,
"rewards/question_recreation_reward_func/mean": 0.04713628068566322,
"rewards/question_recreation_reward_func/std": 0.032215227372944355,
"rewards/soft_format_reward_func/mean": 0.0,
"rewards/soft_format_reward_func/std": 0.0,
"rewards/strict_format_reward_func/mean": 0.0,
"rewards/strict_format_reward_func/std": 0.0,
"rewards/xmlcount_reward_func/mean": 0.20775000005960464,
"rewards/xmlcount_reward_func/std": 0.2513107433915138,
"step": 6
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 198.0,
"completions/max_terminated_length": 98.5,
"completions/mean_length": 95.5,
"completions/mean_terminated_length": 69.5,
"completions/min_length": 44.5,
"completions/min_terminated_length": 44.5,
"epoch": 0.6956521739130435,
"frac_reward_zero_std": 0.0,
"grad_norm": 19.808725357055664,
"kl": 0.015141080017201602,
"learning_rate": 1.2500000000000005e-07,
"loss": 0.0002,
"num_tokens": 6528.0,
"reward": 0.2949913889169693,
"reward_std": 0.13783530704677105,
"rewards/concensus_correctness_reward_func/mean": 0.0,
"rewards/concensus_correctness_reward_func/std": 0.0,
"rewards/consensus_reward_func/mean": 0.0,
"rewards/consensus_reward_func/std": 0.0,
"rewards/cumulative_reward_2/mean": 0.0,
"rewards/cumulative_reward_2/std": 0.0,
"rewards/final_correctness_reward_func/mean": 0.0,
"rewards/final_correctness_reward_func/std": 0.0,
"rewards/question_recreation_reward_func/mean": 0.04011638090014458,
"rewards/question_recreation_reward_func/std": 0.03736674599349499,
"rewards/soft_format_reward_func/mean": 0.0,
"rewards/soft_format_reward_func/std": 0.0,
"rewards/strict_format_reward_func/mean": 0.0,
"rewards/strict_format_reward_func/std": 0.0,
"rewards/xmlcount_reward_func/mean": 0.2548750042915344,
"rewards/xmlcount_reward_func/std": 0.30075830966234207,
"step": 8
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 63.5,
"completions/max_terminated_length": 63.5,
"completions/mean_length": 49.5,
"completions/mean_terminated_length": 49.5,
"completions/min_length": 24.5,
"completions/min_terminated_length": 24.5,
"epoch": 0.8695652173913043,
"frac_reward_zero_std": 0.0,
"grad_norm": 43.93717956542969,
"kl": 0.027774153277277946,
"learning_rate": 1.507684480352292e-08,
"loss": -0.0862,
"num_tokens": 7948.0,
"reward": 0.4353444501757622,
"reward_std": 0.1529897004365921,
"rewards/concensus_correctness_reward_func/mean": 0.0,
"rewards/concensus_correctness_reward_func/std": 0.0,
"rewards/consensus_reward_func/mean": 0.0,
"rewards/consensus_reward_func/std": 0.0,
"rewards/cumulative_reward_2/mean": 0.0,
"rewards/cumulative_reward_2/std": 0.0,
"rewards/final_correctness_reward_func/mean": 0.0,
"rewards/final_correctness_reward_func/std": 0.0,
"rewards/question_recreation_reward_func/mean": 0.132094481959939,
"rewards/question_recreation_reward_func/std": 0.03221887769177556,
"rewards/soft_format_reward_func/mean": 0.0,
"rewards/soft_format_reward_func/std": 0.0,
"rewards/strict_format_reward_func/mean": 0.0,
"rewards/strict_format_reward_func/std": 0.0,
"rewards/xmlcount_reward_func/mean": 0.3032499924302101,
"rewards/xmlcount_reward_func/std": 0.16571738570928574,
"step": 10
},
{
"epoch": 0.8695652173913043,
"step": 10,
"total_flos": 0.0,
"train_loss": 0.019416041672229767,
"train_runtime": 1276.5475,
"train_samples_per_second": 0.031,
"train_steps_per_second": 0.008
}
],
"logging_steps": 2,
"max_steps": 10,
"num_input_tokens_seen": 7948,
"num_train_epochs": 1,
"save_steps": 10,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": true
},
"attributes": {}
}
},
"total_flos": 0.0,
"train_batch_size": 2,
"trial_name": null,
"trial_params": null
}