{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 3.0, "eval_steps": 500, "global_step": 519, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.057971014492753624, "grad_norm": 29.952982357696865, "learning_rate": 1.7307692307692308e-06, "loss": 2.009, "step": 10 }, { "epoch": 0.11594202898550725, "grad_norm": 11.512139435162283, "learning_rate": 3.653846153846154e-06, "loss": 1.8239, "step": 20 }, { "epoch": 0.17391304347826086, "grad_norm": 6.626968877090522, "learning_rate": 5.576923076923077e-06, "loss": 1.1081, "step": 30 }, { "epoch": 0.2318840579710145, "grad_norm": 5.700821840146368, "learning_rate": 7.500000000000001e-06, "loss": 0.4989, "step": 40 }, { "epoch": 0.2898550724637681, "grad_norm": 3.4089761833344654, "learning_rate": 9.423076923076923e-06, "loss": 0.174, "step": 50 }, { "epoch": 0.34782608695652173, "grad_norm": 0.5938704052704902, "learning_rate": 9.994457294322858e-06, "loss": 0.0671, "step": 60 }, { "epoch": 0.4057971014492754, "grad_norm": 0.3578679075693625, "learning_rate": 9.967338926801066e-06, "loss": 0.0486, "step": 70 }, { "epoch": 0.463768115942029, "grad_norm": 0.27829454373593665, "learning_rate": 9.917749373597506e-06, "loss": 0.0374, "step": 80 }, { "epoch": 0.5217391304347826, "grad_norm": 0.41774241578480886, "learning_rate": 9.84591296731245e-06, "loss": 0.0304, "step": 90 }, { "epoch": 0.5797101449275363, "grad_norm": 0.7351453111684326, "learning_rate": 9.752154680581783e-06, "loss": 0.0313, "step": 100 }, { "epoch": 0.6376811594202898, "grad_norm": 0.6835184704362615, "learning_rate": 9.636898655969837e-06, "loss": 0.0259, "step": 110 }, { "epoch": 0.6956521739130435, "grad_norm": 0.7320159996448656, "learning_rate": 9.500666287238573e-06, "loss": 0.0267, "step": 120 }, { "epoch": 0.7536231884057971, "grad_norm": 0.9706818562248596, "learning_rate": 9.344073860673016e-06, "loss": 0.022, "step": 130 }, { "epoch": 0.8115942028985508, "grad_norm": 1.3747459132928828, "learning_rate": 9.167829767133047e-06, "loss": 0.0211, "step": 140 }, { "epoch": 0.8695652173913043, "grad_norm": 0.3625030094557575, "learning_rate": 8.972731297443722e-06, "loss": 0.018, "step": 150 }, { "epoch": 0.927536231884058, "grad_norm": 0.32466805266951876, "learning_rate": 8.759661035620992e-06, "loss": 0.0151, "step": 160 }, { "epoch": 0.9855072463768116, "grad_norm": 0.3393035460364341, "learning_rate": 8.529582866249187e-06, "loss": 0.0139, "step": 170 }, { "epoch": 1.0405797101449274, "grad_norm": 0.3997122346830682, "learning_rate": 8.283537614071987e-06, "loss": 0.0113, "step": 180 }, { "epoch": 1.098550724637681, "grad_norm": 2.3217506351085917, "learning_rate": 8.022638335522484e-06, "loss": 0.0095, "step": 190 }, { "epoch": 1.1565217391304348, "grad_norm": 0.3656786470769303, "learning_rate": 7.748065283492397e-06, "loss": 0.0084, "step": 200 }, { "epoch": 1.2144927536231884, "grad_norm": 0.2974541884546707, "learning_rate": 7.461060568118822e-06, "loss": 0.0082, "step": 210 }, { "epoch": 1.272463768115942, "grad_norm": 0.3621483810363318, "learning_rate": 7.162922537741937e-06, "loss": 0.0079, "step": 220 }, { "epoch": 1.3304347826086955, "grad_norm": 0.4604114786251555, "learning_rate": 6.854999905453022e-06, "loss": 0.0069, "step": 230 }, { "epoch": 1.3884057971014494, "grad_norm": 0.4826645672344626, "learning_rate": 6.538685647803049e-06, "loss": 0.0058, "step": 240 }, { "epoch": 1.4463768115942028, "grad_norm": 0.28726402847877164, "learning_rate": 6.215410703272805e-06, "loss": 0.0053, "step": 250 }, { "epoch": 1.5043478260869565, "grad_norm": 0.1648274095907742, "learning_rate": 5.8866374990112785e-06, "loss": 0.0047, "step": 260 }, { "epoch": 1.5623188405797102, "grad_norm": 0.832116359446373, "learning_rate": 5.5538533351260395e-06, "loss": 0.0049, "step": 270 }, { "epoch": 1.6202898550724638, "grad_norm": 0.35204832433816197, "learning_rate": 5.218563656453609e-06, "loss": 0.004, "step": 280 }, { "epoch": 1.6782608695652175, "grad_norm": 0.5085831024000373, "learning_rate": 4.882285242246958e-06, "loss": 0.0035, "step": 290 }, { "epoch": 1.736231884057971, "grad_norm": 0.5674575793970932, "learning_rate": 4.546539344588486e-06, "loss": 0.003, "step": 300 }, { "epoch": 1.7942028985507248, "grad_norm": 0.32551366582321456, "learning_rate": 4.212844806568906e-06, "loss": 0.003, "step": 310 }, { "epoch": 1.8521739130434782, "grad_norm": 0.18906388500680382, "learning_rate": 3.88271119136386e-06, "loss": 0.0022, "step": 320 }, { "epoch": 1.9101449275362319, "grad_norm": 0.5113664546085201, "learning_rate": 3.557631953290914e-06, "loss": 0.0022, "step": 330 }, { "epoch": 1.9681159420289855, "grad_norm": 0.3861190420702106, "learning_rate": 3.239077681739618e-06, "loss": 0.0021, "step": 340 }, { "epoch": 2.0231884057971015, "grad_norm": 0.15897987765525454, "learning_rate": 2.9284894485376057e-06, "loss": 0.0017, "step": 350 }, { "epoch": 2.081159420289855, "grad_norm": 0.6453986830817099, "learning_rate": 2.6272722888479152e-06, "loss": 0.0011, "step": 360 }, { "epoch": 2.139130434782609, "grad_norm": 0.12469102836354488, "learning_rate": 2.336788845088478e-06, "loss": 0.0009, "step": 370 }, { "epoch": 2.197101449275362, "grad_norm": 0.5210826236391908, "learning_rate": 2.058353202627417e-06, "loss": 0.0009, "step": 380 }, { "epoch": 2.255072463768116, "grad_norm": 0.14107015806726528, "learning_rate": 1.7932249451400863e-06, "loss": 0.0007, "step": 390 }, { "epoch": 2.3130434782608695, "grad_norm": 0.17094040993245266, "learning_rate": 1.542603456520214e-06, "loss": 0.0008, "step": 400 }, { "epoch": 2.3710144927536234, "grad_norm": 0.29173388058367744, "learning_rate": 1.3076224951220413e-06, "loss": 0.0008, "step": 410 }, { "epoch": 2.428985507246377, "grad_norm": 0.3029241786417147, "learning_rate": 1.0893450648784736e-06, "loss": 0.0007, "step": 420 }, { "epoch": 2.4869565217391303, "grad_norm": 0.17573523653808565, "learning_rate": 8.887586064971859e-07, "loss": 0.0007, "step": 430 }, { "epoch": 2.544927536231884, "grad_norm": 0.15567558268935872, "learning_rate": 7.067705304887074e-07, "loss": 0.0008, "step": 440 }, { "epoch": 2.6028985507246376, "grad_norm": 0.21209876832693972, "learning_rate": 5.442041122341057e-07, "loss": 0.0007, "step": 450 }, { "epoch": 2.660869565217391, "grad_norm": 0.09285510522354712, "learning_rate": 4.0179476766211865e-07, "loss": 0.0006, "step": 460 }, { "epoch": 2.718840579710145, "grad_norm": 0.13539077575014008, "learning_rate": 2.8018672638378486e-07, "loss": 0.0007, "step": 470 }, { "epoch": 2.776811594202899, "grad_norm": 0.9942412093476435, "learning_rate": 1.7993011733458077e-07, "loss": 0.0006, "step": 480 }, { "epoch": 2.8347826086956522, "grad_norm": 0.13127636159227557, "learning_rate": 1.0147848010803319e-07, "loss": 0.0006, "step": 490 }, { "epoch": 2.8927536231884057, "grad_norm": 0.10614694682566984, "learning_rate": 4.5186713238979385e-08, "loss": 0.0005, "step": 500 }, { "epoch": 2.9507246376811596, "grad_norm": 0.13240069629057752, "learning_rate": 1.1309468718013194e-08, "loss": 0.0007, "step": 510 }, { "epoch": 3.0, "step": 519, "total_flos": 457198286143488.0, "train_loss": 0.11712381789758658, "train_runtime": 11084.2911, "train_samples_per_second": 2.988, "train_steps_per_second": 0.047 } ], "logging_steps": 10, "max_steps": 519, "num_input_tokens_seen": 0, "num_train_epochs": 3, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 457198286143488.0, "train_batch_size": 2, "trial_name": null, "trial_params": null }