{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 10.0, "eval_steps": 500, "global_step": 10, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 47.5, "completions/max_terminated_length": 47.5, "completions/mean_length": 41.375, "completions/mean_terminated_length": 41.375, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "epoch": 2.0, "frac_reward_zero_std": 0.75, "grad_norm": 0.0, "kl": 0.0, "learning_rate": 5e-07, "loss": -0.0063, "num_tokens": 1355.0, "reward": 0.5854154527187347, "reward_std": 0.0009854354429990053, "rewards/concensus_correctness_reward_func/mean": 0.0, "rewards/concensus_correctness_reward_func/std": 0.0, "rewards/consensus_reward_func/mean": 0.0, "rewards/consensus_reward_func/std": 0.0, "rewards/cumulative_reward_2/mean": 0.0, "rewards/cumulative_reward_2/std": 0.0, "rewards/final_correctness_reward_func/mean": 0.0, "rewards/final_correctness_reward_func/std": 0.0, "rewards/question_recreation_reward_func/mean": 0.08541545830667019, "rewards/question_recreation_reward_func/std": 0.057605274370871484, "rewards/soft_format_reward_func/mean": 0.0, "rewards/soft_format_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.0, "rewards/strict_format_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 2 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 48.0, "completions/max_terminated_length": 48.0, "completions/mean_length": 39.25, "completions/mean_terminated_length": 39.25, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "epoch": 4.0, "frac_reward_zero_std": 0.5, "grad_norm": 9.636852264404297, "kl": 0.02098847759043565, "learning_rate": 4.415111107797445e-07, "loss": -0.0376, "num_tokens": 2693.0, "reward": 0.47920550405979156, "reward_std": 0.08743657803279348, "rewards/concensus_correctness_reward_func/mean": 0.0, "rewards/concensus_correctness_reward_func/std": 0.0, "rewards/consensus_reward_func/mean": 0.0, "rewards/consensus_reward_func/std": 0.0, "rewards/cumulative_reward_2/mean": 0.0, "rewards/cumulative_reward_2/std": 0.0, "rewards/final_correctness_reward_func/mean": 0.0, "rewards/final_correctness_reward_func/std": 0.0, "rewards/question_recreation_reward_func/mean": 0.041955491527915, "rewards/question_recreation_reward_func/std": 0.00228323187911883, "rewards/soft_format_reward_func/mean": 0.0, "rewards/soft_format_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.0, "rewards/strict_format_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4372500032186508, "rewards/xmlcount_reward_func/std": 0.12550000101327896, "step": 4 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 49.0, "completions/max_terminated_length": 49.0, "completions/mean_length": 42.5, "completions/mean_terminated_length": 42.5, "completions/min_length": 38.5, "completions/min_terminated_length": 38.5, "epoch": 6.0, "frac_reward_zero_std": 0.5, "grad_norm": 10.108943939208984, "kl": 0.011130992257676553, "learning_rate": 2.934120444167326e-07, "loss": 0.0088, "num_tokens": 4057.0, "reward": 0.5809410214424133, "reward_std": 0.005224586755502969, "rewards/concensus_correctness_reward_func/mean": 0.0, "rewards/concensus_correctness_reward_func/std": 0.0, "rewards/consensus_reward_func/mean": 0.0, "rewards/consensus_reward_func/std": 0.0, "rewards/cumulative_reward_2/mean": 0.0, "rewards/cumulative_reward_2/std": 0.0, "rewards/final_correctness_reward_func/mean": 0.0, "rewards/final_correctness_reward_func/std": 0.0, "rewards/question_recreation_reward_func/mean": 0.08094101399183273, "rewards/question_recreation_reward_func/std": 0.0538091630442068, "rewards/soft_format_reward_func/mean": 0.0, "rewards/soft_format_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.0, "rewards/strict_format_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 6 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 70.5, "completions/max_terminated_length": 70.5, "completions/mean_length": 48.875, "completions/mean_terminated_length": 48.875, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "epoch": 8.0, "frac_reward_zero_std": 0.5, "grad_norm": 0.0004040227795485407, "kl": 0.017657065225648694, "learning_rate": 1.2500000000000005e-07, "loss": 0.0664, "num_tokens": 5472.0, "reward": 0.475370854139328, "reward_std": 0.0956975594162941, "rewards/concensus_correctness_reward_func/mean": 0.0, "rewards/concensus_correctness_reward_func/std": 0.0, "rewards/consensus_reward_func/mean": 0.0, "rewards/consensus_reward_func/std": 0.0, "rewards/cumulative_reward_2/mean": 0.0, "rewards/cumulative_reward_2/std": 0.0, "rewards/final_correctness_reward_func/mean": 0.0, "rewards/final_correctness_reward_func/std": 0.0, "rewards/question_recreation_reward_func/mean": 0.04062086530029774, "rewards/question_recreation_reward_func/std": 0.004772845219122246, "rewards/soft_format_reward_func/mean": 0.0, "rewards/soft_format_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.0, "rewards/strict_format_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4347499907016754, "rewards/xmlcount_reward_func/std": 0.13050000369548798, "step": 8 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 44.0, "completions/max_terminated_length": 44.0, "completions/mean_length": 41.0, "completions/mean_terminated_length": 41.0, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "epoch": 10.0, "frac_reward_zero_std": 0.75, "grad_norm": 8.019640922546387, "kl": 0.001891768868517829, "learning_rate": 1.507684480352292e-08, "loss": 0.0, "num_tokens": 6824.0, "reward": 0.5831517577171326, "reward_std": 0.0021489623468369246, "rewards/concensus_correctness_reward_func/mean": 0.0, "rewards/concensus_correctness_reward_func/std": 0.0, "rewards/consensus_reward_func/mean": 0.0, "rewards/consensus_reward_func/std": 0.0, "rewards/cumulative_reward_2/mean": 0.0, "rewards/cumulative_reward_2/std": 0.0, "rewards/final_correctness_reward_func/mean": 0.0, "rewards/final_correctness_reward_func/std": 0.0, "rewards/question_recreation_reward_func/mean": 0.08315177075564861, "rewards/question_recreation_reward_func/std": 0.05820586674963124, "rewards/soft_format_reward_func/mean": 0.0, "rewards/soft_format_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.0, "rewards/strict_format_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 10 }, { "epoch": 10.0, "step": 10, "total_flos": 0.0, "train_loss": 0.006285320222377777, "train_runtime": 1166.67, "train_samples_per_second": 0.034, "train_steps_per_second": 0.009 } ], "logging_steps": 2, "max_steps": 10, "num_input_tokens_seen": 6824, "num_train_epochs": 10, "save_steps": 10, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 0.0, "train_batch_size": 2, "trial_name": null, "trial_params": null }