{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 3.0, "eval_steps": 500, "global_step": 3477, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.08628127696289906, "grad_norm": 1.6037696599960327, "learning_rate": 4.857635893011217e-05, "loss": 8.298780517578125, "step": 100 }, { "epoch": 0.1725625539257981, "grad_norm": 1.7524189949035645, "learning_rate": 4.7138337647397185e-05, "loss": 7.087490844726562, "step": 200 }, { "epoch": 0.25884383088869717, "grad_norm": 1.324486494064331, "learning_rate": 4.57003163646822e-05, "loss": 6.956798095703125, "step": 300 }, { "epoch": 0.3451251078515962, "grad_norm": 1.4535937309265137, "learning_rate": 4.426229508196721e-05, "loss": 6.822440795898437, "step": 400 }, { "epoch": 0.4314063848144953, "grad_norm": 1.327022910118103, "learning_rate": 4.282427379925223e-05, "loss": 6.734780883789062, "step": 500 }, { "epoch": 0.5176876617773943, "grad_norm": 1.1214697360992432, "learning_rate": 4.138625251653725e-05, "loss": 6.682239379882812, "step": 600 }, { "epoch": 0.6039689387402933, "grad_norm": 1.5475066900253296, "learning_rate": 3.994823123382226e-05, "loss": 6.600054321289062, "step": 700 }, { "epoch": 0.6902502157031924, "grad_norm": 1.3128082752227783, "learning_rate": 3.8510209951107276e-05, "loss": 6.516415405273437, "step": 800 }, { "epoch": 0.7765314926660914, "grad_norm": 1.6903055906295776, "learning_rate": 3.707218866839229e-05, "loss": 6.505012817382813, "step": 900 }, { "epoch": 0.8628127696289906, "grad_norm": 1.616003394126892, "learning_rate": 3.563416738567731e-05, "loss": 6.397700805664062, "step": 1000 }, { "epoch": 0.9490940465918896, "grad_norm": 1.7408578395843506, "learning_rate": 3.4196146102962325e-05, "loss": 6.376287841796875, "step": 1100 }, { "epoch": 1.0353753235547887, "grad_norm": 1.6899269819259644, "learning_rate": 3.2758124820247346e-05, "loss": 6.3477630615234375, "step": 1200 }, { "epoch": 1.1216566005176876, "grad_norm": 1.5354853868484497, "learning_rate": 3.132010353753236e-05, "loss": 6.22337158203125, "step": 1300 }, { "epoch": 1.2079378774805867, "grad_norm": 1.530237078666687, "learning_rate": 2.9882082254817374e-05, "loss": 6.231620483398437, "step": 1400 }, { "epoch": 1.2942191544434858, "grad_norm": 1.71237313747406, "learning_rate": 2.8444060972102388e-05, "loss": 6.191749877929688, "step": 1500 }, { "epoch": 1.380500431406385, "grad_norm": 1.532090425491333, "learning_rate": 2.7006039689387402e-05, "loss": 6.148622436523437, "step": 1600 }, { "epoch": 1.4667817083692838, "grad_norm": 1.5732359886169434, "learning_rate": 2.5568018406672423e-05, "loss": 6.130806274414063, "step": 1700 }, { "epoch": 1.553062985332183, "grad_norm": 1.74668288230896, "learning_rate": 2.4129997123957434e-05, "loss": 6.11656005859375, "step": 1800 }, { "epoch": 1.639344262295082, "grad_norm": 1.6298942565917969, "learning_rate": 2.269197584124245e-05, "loss": 6.120827026367188, "step": 1900 }, { "epoch": 1.725625539257981, "grad_norm": 1.5918290615081787, "learning_rate": 2.125395455852747e-05, "loss": 6.088132934570313, "step": 2000 }, { "epoch": 1.8119068162208802, "grad_norm": 1.4154537916183472, "learning_rate": 1.9815933275812482e-05, "loss": 6.068784790039063, "step": 2100 }, { "epoch": 1.8981880931837791, "grad_norm": 1.5682926177978516, "learning_rate": 1.83779119930975e-05, "loss": 6.038920288085937, "step": 2200 }, { "epoch": 1.984469370146678, "grad_norm": 1.5493305921554565, "learning_rate": 1.6939890710382514e-05, "loss": 6.069011840820313, "step": 2300 }, { "epoch": 2.0707506471095773, "grad_norm": 1.7473928928375244, "learning_rate": 1.550186942766753e-05, "loss": 6.010381469726562, "step": 2400 }, { "epoch": 2.1570319240724762, "grad_norm": 1.7771306037902832, "learning_rate": 1.4063848144952545e-05, "loss": 5.94528076171875, "step": 2500 }, { "epoch": 2.243313201035375, "grad_norm": 1.8056012392044067, "learning_rate": 1.2625826862237561e-05, "loss": 5.949163208007812, "step": 2600 }, { "epoch": 2.3295944779982745, "grad_norm": 1.547564148902893, "learning_rate": 1.1187805579522577e-05, "loss": 5.962681884765625, "step": 2700 }, { "epoch": 2.4158757549611733, "grad_norm": 1.4664490222930908, "learning_rate": 9.749784296807593e-06, "loss": 5.9176702880859375, "step": 2800 }, { "epoch": 2.5021570319240727, "grad_norm": 1.5322167873382568, "learning_rate": 8.31176301409261e-06, "loss": 5.913838500976563, "step": 2900 }, { "epoch": 2.5884383088869716, "grad_norm": 1.51213538646698, "learning_rate": 6.873741731377626e-06, "loss": 5.912578735351563, "step": 3000 }, { "epoch": 2.6747195858498705, "grad_norm": 1.7505192756652832, "learning_rate": 5.435720448662641e-06, "loss": 5.9252081298828125, "step": 3100 }, { "epoch": 2.76100086281277, "grad_norm": 1.5077073574066162, "learning_rate": 3.997699165947656e-06, "loss": 5.895281982421875, "step": 3200 }, { "epoch": 2.8472821397756687, "grad_norm": 1.7371774911880493, "learning_rate": 2.559677883232672e-06, "loss": 5.896738891601562, "step": 3300 }, { "epoch": 2.9335634167385676, "grad_norm": 1.5654444694519043, "learning_rate": 1.1216566005176878e-06, "loss": 5.89673583984375, "step": 3400 }, { "epoch": 3.0, "step": 3477, "total_flos": 3634049581056000.0, "train_loss": 6.285199090465582, "train_runtime": 228.5387, "train_samples_per_second": 30.428, "train_steps_per_second": 15.214 } ], "logging_steps": 100, "max_steps": 3477, "num_input_tokens_seen": 0, "num_train_epochs": 3, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 3634049581056000.0, "train_batch_size": 2, "trial_name": null, "trial_params": null }