{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 0.802937576499388, "eval_steps": 500, "global_step": 328, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.02447980416156671, "grad_norm": 6.0, "learning_rate": 4.5e-07, "loss": 2.0334224700927734, "step": 10 }, { "epoch": 0.04895960832313342, "grad_norm": 4.9375, "learning_rate": 4.990124606538042e-07, "loss": 2.123665428161621, "step": 20 }, { "epoch": 0.07343941248470012, "grad_norm": 13.875, "learning_rate": 4.956087596148824e-07, "loss": 2.0373281478881835, "step": 30 }, { "epoch": 0.09791921664626684, "grad_norm": 12.3125, "learning_rate": 4.898098898861766e-07, "loss": 2.0780839920043945, "step": 40 }, { "epoch": 0.12239902080783353, "grad_norm": 11.6875, "learning_rate": 4.816724018579583e-07, "loss": 2.168250846862793, "step": 50 }, { "epoch": 0.14687882496940025, "grad_norm": 27.5, "learning_rate": 4.7127565205061093e-07, "loss": 2.0774251937866213, "step": 60 }, { "epoch": 0.17135862913096694, "grad_norm": 14.6875, "learning_rate": 4.587210292324061e-07, "loss": 2.2656288146972656, "step": 70 }, { "epoch": 0.19583843329253367, "grad_norm": 13.3125, "learning_rate": 4.441309656795106e-07, "loss": 1.990157699584961, "step": 80 }, { "epoch": 0.22031823745410037, "grad_norm": 5.3125, "learning_rate": 4.2764774322038486e-07, "loss": 2.2658514022827148, "step": 90 }, { "epoch": 0.24479804161566707, "grad_norm": 24.375, "learning_rate": 4.094321057079873e-07, "loss": 2.0881725311279298, "step": 100 }, { "epoch": 0.2692778457772338, "grad_norm": 5.5625, "learning_rate": 3.896616914509131e-07, "loss": 2.186654281616211, "step": 110 }, { "epoch": 0.2937576499388005, "grad_norm": 15.5, "learning_rate": 3.6852930089034707e-07, "loss": 2.238821601867676, "step": 120 }, { "epoch": 0.3182374541003672, "grad_norm": 14.1875, "learning_rate": 3.4624101641638926e-07, "loss": 2.05932559967041, "step": 130 }, { "epoch": 0.3427172582619339, "grad_norm": 10.875, "learning_rate": 3.2301419265924393e-07, "loss": 2.070750617980957, "step": 140 }, { "epoch": 0.3671970624235006, "grad_norm": 10.75, "learning_rate": 2.9907533685388716e-07, "loss": 2.12088565826416, "step": 150 }, { "epoch": 0.39167686658506734, "grad_norm": 5.15625, "learning_rate": 2.7465789994882794e-07, "loss": 1.9712303161621094, "step": 160 }, { "epoch": 0.41615667074663404, "grad_norm": 9.625, "learning_rate": 2.5e-07, "loss": 2.1383413314819335, "step": 170 }, { "epoch": 0.44063647490820074, "grad_norm": 7.34375, "learning_rate": 2.2534210005117207e-07, "loss": 2.009996032714844, "step": 180 }, { "epoch": 0.46511627906976744, "grad_norm": 10.0, "learning_rate": 2.0092466314611287e-07, "loss": 2.076437759399414, "step": 190 }, { "epoch": 0.48959608323133413, "grad_norm": 9.6875, "learning_rate": 1.7698580734075607e-07, "loss": 2.1122467041015627, "step": 200 }, { "epoch": 0.5140758873929009, "grad_norm": 7.53125, "learning_rate": 1.5375898358361077e-07, "loss": 1.956191062927246, "step": 210 }, { "epoch": 0.5385556915544676, "grad_norm": 10.3125, "learning_rate": 1.3147069910965296e-07, "loss": 1.944639015197754, "step": 220 }, { "epoch": 0.5630354957160343, "grad_norm": 13.3125, "learning_rate": 1.1033830854908691e-07, "loss": 2.2447561264038085, "step": 230 }, { "epoch": 0.587515299877601, "grad_norm": 11.1875, "learning_rate": 9.05678942920127e-08, "loss": 2.1660913467407226, "step": 240 }, { "epoch": 0.6119951040391677, "grad_norm": 18.5, "learning_rate": 7.235225677961512e-08, "loss": 2.3716936111450195, "step": 250 }, { "epoch": 0.6364749082007344, "grad_norm": 8.3125, "learning_rate": 5.586903432048942e-08, "loss": 2.1261455535888674, "step": 260 }, { "epoch": 0.6609547123623011, "grad_norm": 8.8125, "learning_rate": 4.127897076759399e-08, "loss": 2.064971923828125, "step": 270 }, { "epoch": 0.6854345165238678, "grad_norm": 14.5625, "learning_rate": 2.8724347949389048e-08, "loss": 2.085330009460449, "step": 280 }, { "epoch": 0.7099143206854345, "grad_norm": 17.25, "learning_rate": 1.8327598142041656e-08, "loss": 2.1130571365356445, "step": 290 }, { "epoch": 0.7343941248470012, "grad_norm": 10.625, "learning_rate": 1.0190110113823425e-08, "loss": 2.056776428222656, "step": 300 }, { "epoch": 0.758873929008568, "grad_norm": 13.75, "learning_rate": 4.391240385117623e-09, "loss": 2.13861141204834, "step": 310 }, { "epoch": 0.7833537331701347, "grad_norm": 14.8125, "learning_rate": 9.87539346195776e-10, "loss": 2.129196357727051, "step": 320 }, { "epoch": 0.802937576499388, "step": 328, "total_flos": 1.5578264772983194e+17, "train_loss": 2.1087728360804117, "train_runtime": 2487.2809, "train_samples_per_second": 0.526, "train_steps_per_second": 0.132 } ], "logging_steps": 10, "max_steps": 328, "num_input_tokens_seen": 0, "num_train_epochs": 1, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": false, "should_training_stop": false }, "attributes": {} } }, "total_flos": 1.5578264772983194e+17, "train_batch_size": 1, "trial_name": null, "trial_params": null }