{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 1.6493055555555556, "eval_steps": 500, "global_step": 950, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.034722222222222224, "grad_norm": 3.1068832874298096, "learning_rate": 1.6379310344827587e-06, "loss": 2.675, "step": 20 }, { "epoch": 0.06944444444444445, "grad_norm": 2.46288800239563, "learning_rate": 3.362068965517242e-06, "loss": 2.4759, "step": 40 }, { "epoch": 0.10416666666666667, "grad_norm": 2.1140520572662354, "learning_rate": 5.086206896551724e-06, "loss": 2.3011, "step": 60 }, { "epoch": 0.1388888888888889, "grad_norm": 1.9008077383041382, "learning_rate": 6.810344827586207e-06, "loss": 2.0724, "step": 80 }, { "epoch": 0.1736111111111111, "grad_norm": 2.368029832839966, "learning_rate": 8.53448275862069e-06, "loss": 1.9312, "step": 100 }, { "epoch": 0.20833333333333334, "grad_norm": 2.3201775550842285, "learning_rate": 9.999793100349294e-06, "loss": 1.8941, "step": 120 }, { "epoch": 0.24305555555555555, "grad_norm": 2.499159812927246, "learning_rate": 9.987843743451796e-06, "loss": 1.7747, "step": 140 }, { "epoch": 0.2777777777777778, "grad_norm": 2.1269280910491943, "learning_rate": 9.957553516178782e-06, "loss": 1.7052, "step": 160 }, { "epoch": 0.3125, "grad_norm": 2.290123224258423, "learning_rate": 9.909033799150947e-06, "loss": 1.7429, "step": 180 }, { "epoch": 0.3472222222222222, "grad_norm": 2.3117945194244385, "learning_rate": 9.842463004902127e-06, "loss": 1.6598, "step": 200 }, { "epoch": 0.3819444444444444, "grad_norm": 2.6141538619995117, "learning_rate": 9.758085921836076e-06, "loss": 1.6429, "step": 220 }, { "epoch": 0.4166666666666667, "grad_norm": 2.4996819496154785, "learning_rate": 9.656212814111567e-06, "loss": 1.6566, "step": 240 }, { "epoch": 0.4513888888888889, "grad_norm": 2.8781185150146484, "learning_rate": 9.53721828076571e-06, "loss": 1.6255, "step": 260 }, { "epoch": 0.4861111111111111, "grad_norm": 2.8994524478912354, "learning_rate": 9.401539878270545e-06, "loss": 1.6064, "step": 280 }, { "epoch": 0.5208333333333334, "grad_norm": 2.9606640338897705, "learning_rate": 9.249676511588e-06, "loss": 1.6141, "step": 300 }, { "epoch": 0.5555555555555556, "grad_norm": 2.9091055393218994, "learning_rate": 9.082186599639429e-06, "loss": 1.598, "step": 320 }, { "epoch": 0.5902777777777778, "grad_norm": 2.523542642593384, "learning_rate": 8.899686021935554e-06, "loss": 1.5779, "step": 340 }, { "epoch": 0.625, "grad_norm": 2.5727219581604004, "learning_rate": 8.702845853917242e-06, "loss": 1.5475, "step": 360 }, { "epoch": 0.6597222222222222, "grad_norm": 2.4910247325897217, "learning_rate": 8.492389899334572e-06, "loss": 1.5346, "step": 380 }, { "epoch": 0.6944444444444444, "grad_norm": 3.080043077468872, "learning_rate": 8.269092028737885e-06, "loss": 1.5282, "step": 400 }, { "epoch": 0.7291666666666666, "grad_norm": 3.0990397930145264, "learning_rate": 8.033773333867498e-06, "loss": 1.4993, "step": 420 }, { "epoch": 0.7638888888888888, "grad_norm": 2.7828781604766846, "learning_rate": 7.78729910840572e-06, "loss": 1.4994, "step": 440 }, { "epoch": 0.7986111111111112, "grad_norm": 3.1256909370422363, "learning_rate": 7.530575666193283e-06, "loss": 1.5023, "step": 460 }, { "epoch": 0.8333333333333334, "grad_norm": 3.1738061904907227, "learning_rate": 7.26454700860997e-06, "loss": 1.4671, "step": 480 }, { "epoch": 0.8680555555555556, "grad_norm": 3.4600017070770264, "learning_rate": 6.990191353373876e-06, "loss": 1.466, "step": 500 }, { "epoch": 0.9027777777777778, "grad_norm": 3.556324005126953, "learning_rate": 6.708517537523264e-06, "loss": 1.4407, "step": 520 }, { "epoch": 0.9375, "grad_norm": 3.1536033153533936, "learning_rate": 6.420561307807713e-06, "loss": 1.4485, "step": 540 }, { "epoch": 0.9722222222222222, "grad_norm": 3.01642107963562, "learning_rate": 6.12738151212918e-06, "loss": 1.4672, "step": 560 }, { "epoch": 1.0069444444444444, "grad_norm": 3.864258050918579, "learning_rate": 5.830056206037482e-06, "loss": 1.4258, "step": 580 }, { "epoch": 1.0416666666666667, "grad_norm": 2.967533826828003, "learning_rate": 5.529678688597081e-06, "loss": 1.3619, "step": 600 }, { "epoch": 1.0763888888888888, "grad_norm": 3.2055039405822754, "learning_rate": 5.2273534822017105e-06, "loss": 1.3592, "step": 620 }, { "epoch": 1.1111111111111112, "grad_norm": 3.8909292221069336, "learning_rate": 4.924192271119554e-06, "loss": 1.3359, "step": 640 }, { "epoch": 1.1458333333333333, "grad_norm": 3.0267176628112793, "learning_rate": 4.621309813703385e-06, "loss": 1.3424, "step": 660 }, { "epoch": 1.1805555555555556, "grad_norm": 4.415727138519287, "learning_rate": 4.319819843296952e-06, "loss": 1.3435, "step": 680 }, { "epoch": 1.2152777777777777, "grad_norm": 3.6791961193084717, "learning_rate": 4.020830972910433e-06, "loss": 1.3391, "step": 700 }, { "epoch": 1.25, "grad_norm": 3.3228704929351807, "learning_rate": 3.7254426187239567e-06, "loss": 1.3054, "step": 720 }, { "epoch": 1.2847222222222223, "grad_norm": 3.616981267929077, "learning_rate": 3.4347409574088896e-06, "loss": 1.312, "step": 740 }, { "epoch": 1.3194444444444444, "grad_norm": 4.216521739959717, "learning_rate": 3.149794932132331e-06, "loss": 1.3334, "step": 760 }, { "epoch": 1.3541666666666667, "grad_norm": 4.408073425292969, "learning_rate": 2.871652321931161e-06, "loss": 1.317, "step": 780 }, { "epoch": 1.3888888888888888, "grad_norm": 2.9930851459503174, "learning_rate": 2.601335888909005e-06, "loss": 1.3089, "step": 800 }, { "epoch": 1.4236111111111112, "grad_norm": 4.05021858215332, "learning_rate": 2.339839617423318e-06, "loss": 1.2859, "step": 820 }, { "epoch": 1.4583333333333333, "grad_norm": 4.1063079833984375, "learning_rate": 2.0881250590915316e-06, "loss": 1.2947, "step": 840 }, { "epoch": 1.4930555555555556, "grad_norm": 3.8521957397460938, "learning_rate": 1.8471177970560712e-06, "loss": 1.3056, "step": 860 }, { "epoch": 1.5277777777777777, "grad_norm": 3.304344415664673, "learning_rate": 1.6177040425095664e-06, "loss": 1.2695, "step": 880 }, { "epoch": 1.5625, "grad_norm": 3.73410701751709, "learning_rate": 1.40072737599522e-06, "loss": 1.3125, "step": 900 }, { "epoch": 1.5972222222222223, "grad_norm": 3.594968318939209, "learning_rate": 1.196985645464921e-06, "loss": 1.2841, "step": 920 }, { "epoch": 1.6319444444444444, "grad_norm": 4.417556285858154, "learning_rate": 1.0072280325013185e-06, "loss": 1.2878, "step": 940 } ], "logging_steps": 20, "max_steps": 1152, "num_input_tokens_seen": 0, "num_train_epochs": 2, "save_steps": 50, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 1.9311640760991744e+17, "train_batch_size": 4, "trial_name": null, "trial_params": null }