{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 0.5, "eval_steps": 500, "global_step": 8, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "buffer_qsize": 0.0, "clip_ratio/high_max": 3.222041777917184e-05, "clip_ratio/high_mean": 3.222041777917184e-05, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 3.222041777917184e-05, "completions/mean_length": 3876.25, "entropy": 0.9389231279492378, "epoch": 0.25, "forward_time_s": 0.30089330673217773, "generation_tok_per_s": 6915.577392578125, "grad_norm": 0.36549293994903564, "kl": 0.0003301510150777176, "learning_rate": 1e-06, "loss": 0.02711719088256359, "queue_wait_time_s": 2.0328194051980972, "ratio": 1.0000957623124123, "reward": 0.4375, "reward_std": 0.49206146597862244, "rewards/r1_distill_math_reward": 0.4375, "scoring_time_ms": 18.674388885498047, "step": 1, "step_time": 6.262407066999003, "train_seq_len": 4371.125, "training_tok/s": 12972.816542826044, "wait_scoring_ms": 0.20329749584197998, "weight_sync_time_s": 0.14123249053955078 }, { "buffer_qsize": 3.5, "clip_ratio/high_max": 1.6543144738534465e-05, "clip_ratio/high_mean": 1.6543144738534465e-05, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 1.6543144738534465e-05, "completions/mean_length": 6120.375, "entropy": 0.7296525910496712, "epoch": 0.5, "forward_time_s": 0.25667285919189453, "generation_tok_per_s": 13383.16943359375, "grad_norm": 0.2867599427700043, "iteration_time_s": 7.6814478730084375, "kl": 0.00025582036323612556, "learning_rate": 1e-06, "loss": 3.546476364135742e-05, "queue_wait_time_s": 0.3530673235654831, "ratio": 1.0000218078494072, "reward": 0.75, "reward_std": 0.4330126941204071, "rewards/r1_distill_math_reward": 0.75, "scoring_time_ms": 5.158496618270874, "step": 2, "step_time": 7.5191129069717135, "train_seq_len": 6374.125, "training_tok/s": 13318.513123161385, "wait_scoring_ms": 5.808052182197571 }, { "buffer_qsize": 18.0, "clip_ratio/high_max": 7.113443098205607e-05, "clip_ratio/high_mean": 7.113443098205607e-05, "clip_ratio/low_mean": 1.5646326573914848e-05, "clip_ratio/low_min": 1.5646326573914848e-05, "clip_ratio/region_mean": 8.678075755597092e-05, "completions/mean_length": 6053.8125, "entropy": 0.8014494031667709, "epoch": 0.75, "forward_time_s": 0.2557235360145569, "generation_tok_per_s": 13382.43359375, "grad_norm": 0.3024653494358063, "iteration_time_s": 7.571961241017561, "kl": 0.00028099300106987357, "learning_rate": 1e-06, "loss": 0.00955304503440857, "queue_wait_time_s": 0.00046597421169281006, "ratio": 1.0000539049506187, "reward": 0.5625, "reward_std": 0.38186579942703247, "rewards/r1_distill_math_reward": 0.5625, "scoring_time_ms": 0.7072950005531311, "step": 3, "step_time": 7.4889817090006545, "train_seq_len": 6390.375, "training_tok/s": 13393.731171849211, "wait_scoring_ms": 12.160233974456787 }, { "buffer_qsize": 34.0, "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 4.13285370086669e-05, "clip_ratio/low_min": 4.13285370086669e-05, "clip_ratio/region_mean": 4.13285370086669e-05, "completions/mean_length": 6813.625, "entropy": 0.811307743191719, "epoch": 1.0, "forward_time_s": 0.2839837670326233, "generation_tok_per_s": 13381.1396484375, "grad_norm": 0.2735491394996643, "iteration_time_s": 8.428669046988944, "kl": 0.00027251767460256815, "learning_rate": 1e-06, "loss": 0.038605645298957825, "queue_wait_time_s": 0.0004654228687286377, "ratio": 0.9999778047204018, "reward": 0.25, "reward_std": 0.4330126941204071, "rewards/r1_distill_math_reward": 0.25, "scoring_time_ms": 1.8631829619407654, "step": 4, "step_time": 8.344294181006262, "train_seq_len": 7421.5, "training_tok/s": 13408.494449990198, "wait_scoring_ms": 14.634688377380371, "weight_sync_time_s": 0.15913748741149902 }, { "buffer_qsize": 0.0, "clip_ratio/high_max": 1.8846135390049312e-05, "clip_ratio/high_mean": 1.8846135390049312e-05, "clip_ratio/low_mean": 1.011736094369553e-05, "clip_ratio/low_min": 1.011736094369553e-05, "clip_ratio/region_mean": 2.8963496333744843e-05, "completions/mean_length": 5484.0, "entropy": 0.7321677580475807, "epoch": 0.125, "forward_time_s": 0.3462470471858978, "generation_tok_per_s": 9634.451171875, "grad_norm": 0.28255149722099304, "kl": 0.00025452028421568684, "learning_rate": 1e-06, "loss": -0.004137326031923294, "queue_wait_time_s": 1.8160016685724258, "ratio": 1.0000781491398811, "reward": 0.5625, "reward_std": 0.38186579942703247, "rewards/r1_distill_math_reward": 0.5625, "scoring_time_ms": 13.104030728340149, "step": 5, "step_time": 7.895209455047734, "train_seq_len": 5787.375, "training_tok/s": 13242.969307036337, "wait_scoring_ms": 0.1778009943664074, "weight_sync_time_s": 0.13536810874938965 }, { "buffer_qsize": 3.0, "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/mean_length": 6134.3125, "entropy": 0.6220929026603699, "epoch": 0.25, "forward_time_s": 0.26168057322502136, "generation_tok_per_s": 11878.65380859375, "grad_norm": 0.17205984890460968, "iteration_time_s": 7.687154909974197, "kl": 0.00023902823340904433, "learning_rate": 1e-06, "loss": 0.0026915371417999268, "queue_wait_time_s": 0.3177753835916519, "ratio": 0.9999898672103882, "reward": 0.375, "reward_std": 0.21650634706020355, "rewards/r1_distill_math_reward": 0.375, "scoring_time_ms": 8.408703327178955, "step": 6, "step_time": 7.5453005689778365, "train_seq_len": 6619.25, "training_tok/s": 13203.215699232549, "wait_scoring_ms": 0.5662760138511658 }, { "buffer_qsize": 18.0, "clip_ratio/high_max": 5.6621894145791885e-05, "clip_ratio/high_mean": 5.6621894145791885e-05, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 5.6621894145791885e-05, "completions/mean_length": 5713.375, "entropy": 0.8666651099920273, "epoch": 0.375, "forward_time_s": 0.2426280975341797, "generation_tok_per_s": 13567.3173828125, "grad_norm": 0.30514439940452576, "iteration_time_s": 7.157755050022388, "kl": 0.0002972106722154422, "learning_rate": 1e-06, "loss": 0.015170708298683167, "queue_wait_time_s": 0.0004561096429824829, "ratio": 1.0000762566924095, "reward": 0.75, "reward_std": 0.40742091834545135, "rewards/r1_distill_math_reward": 0.75, "scoring_time_ms": 2.2001885175704956, "step": 7, "step_time": 7.077003714046441, "train_seq_len": 5968.5, "training_tok/s": 13234.408437249378, "wait_scoring_ms": 14.30916166305542 }, { "buffer_qsize": 34.0, "clip_ratio/high_max": 9.804690307646524e-06, "clip_ratio/high_mean": 9.804690307646524e-06, "clip_ratio/low_mean": 3.2692873901396524e-05, "clip_ratio/low_min": 3.2692873901396524e-05, "clip_ratio/region_mean": 4.249756420904305e-05, "completions/mean_length": 5408.0625, "entropy": 0.9108624383807182, "epoch": 0.5, "forward_time_s": 0.23351335525512695, "generation_tok_per_s": 13553.60693359375, "grad_norm": 0.3738342225551605, "iteration_time_s": 6.897600525000598, "kl": 0.0003173556542606093, "learning_rate": 1e-06, "loss": 0.0034758001565933228, "queue_wait_time_s": 0.0004056990146636963, "ratio": 0.9998656511306763, "reward": 0.5, "reward_std": 0.4330126941204071, "rewards/r1_distill_math_reward": 0.5, "scoring_time_ms": 18.20103907585144, "step": 8, "step_time": 6.822746234043734, "train_seq_len": 5848.75, "training_tok/s": 13207.699007613326, "wait_scoring_ms": 35.41268348693848, "weight_sync_time_s": 0.23273587226867676 } ], "logging_steps": 1, "max_steps": 8, "num_input_tokens_seen": 0, "num_train_epochs": 9223372036854775807, "save_steps": 4, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 6909415142986752.0, "train_batch_size": 2, "trial_name": null, "trial_params": null }