{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 0.5011027790030878, "eval_steps": 500, "global_step": 284, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.00882223202470225, "grad_norm": 3.421875, "learning_rate": 3.3333333333333333e-06, "loss": 1.5331814765930176, "step": 5 }, { "epoch": 0.0176444640494045, "grad_norm": 3.375, "learning_rate": 4.998563448899413e-06, "loss": 1.5866724014282227, "step": 10 }, { "epoch": 0.02646669607410675, "grad_norm": 5.46875, "learning_rate": 4.989790503518888e-06, "loss": 1.5153743743896484, "step": 15 }, { "epoch": 0.035288928098809, "grad_norm": 4.1875, "learning_rate": 4.973070664146885e-06, "loss": 1.528839683532715, "step": 20 }, { "epoch": 0.044111160123511246, "grad_norm": 3.203125, "learning_rate": 4.9484572970368516e-06, "loss": 1.4991332054138184, "step": 25 }, { "epoch": 0.0529333921482135, "grad_norm": 2.171875, "learning_rate": 4.916028962942763e-06, "loss": 1.497140884399414, "step": 30 }, { "epoch": 0.061755624172915746, "grad_norm": 2.140625, "learning_rate": 4.8758891663695165e-06, "loss": 1.5790274620056153, "step": 35 }, { "epoch": 0.070577856197618, "grad_norm": 2.484375, "learning_rate": 4.828166025208059e-06, "loss": 1.5177926063537597, "step": 40 }, { "epoch": 0.07940008822232025, "grad_norm": 2.5, "learning_rate": 4.773011861809694e-06, "loss": 1.5248671531677247, "step": 45 }, { "epoch": 0.08822232024702249, "grad_norm": 2.25, "learning_rate": 4.710602716804784e-06, "loss": 1.5441174507141113, "step": 50 }, { "epoch": 0.09704455227172475, "grad_norm": 2.140625, "learning_rate": 4.64113778721764e-06, "loss": 1.5332836151123046, "step": 55 }, { "epoch": 0.105866784296427, "grad_norm": 5.46875, "learning_rate": 4.564838790671e-06, "loss": 1.5188942909240724, "step": 60 }, { "epoch": 0.11468901632112924, "grad_norm": 2.78125, "learning_rate": 4.481949257709442e-06, "loss": 1.6403696060180664, "step": 65 }, { "epoch": 0.12351124834583149, "grad_norm": 2.6875, "learning_rate": 4.39273375450049e-06, "loss": 1.4794146537780761, "step": 70 }, { "epoch": 0.13233348037053375, "grad_norm": 2.09375, "learning_rate": 4.297477038394368e-06, "loss": 1.4949344635009765, "step": 75 }, { "epoch": 0.141155712395236, "grad_norm": 2.28125, "learning_rate": 4.196483149037707e-06, "loss": 1.4376185417175293, "step": 80 }, { "epoch": 0.14997794441993825, "grad_norm": 2.484375, "learning_rate": 4.090074437942155e-06, "loss": 1.4953926086425782, "step": 85 }, { "epoch": 0.1588001764446405, "grad_norm": 2.890625, "learning_rate": 3.978590539605338e-06, "loss": 1.5234898567199706, "step": 90 }, { "epoch": 0.16762240846934273, "grad_norm": 2.046875, "learning_rate": 3.862387287468095e-06, "loss": 1.5017153739929199, "step": 95 }, { "epoch": 0.17644464049404499, "grad_norm": 2.875, "learning_rate": 3.741835578168071e-06, "loss": 1.471701431274414, "step": 100 }, { "epoch": 0.18526687251874724, "grad_norm": 2.796875, "learning_rate": 3.6173201877147134e-06, "loss": 1.5574091911315917, "step": 105 }, { "epoch": 0.1940891045434495, "grad_norm": 1.9921875, "learning_rate": 3.4892385433641875e-06, "loss": 1.5162025451660157, "step": 110 }, { "epoch": 0.20291133656815175, "grad_norm": 5.40625, "learning_rate": 3.357999455114148e-06, "loss": 1.356987762451172, "step": 115 }, { "epoch": 0.211733568592854, "grad_norm": 2.484375, "learning_rate": 3.2240218108671683e-06, "loss": 1.489048671722412, "step": 120 }, { "epoch": 0.22055580061755625, "grad_norm": 2.03125, "learning_rate": 3.0877332394275806e-06, "loss": 1.4940855979919434, "step": 125 }, { "epoch": 0.22937803264225848, "grad_norm": 4.0625, "learning_rate": 2.949568745599182e-06, "loss": 1.456082820892334, "step": 130 }, { "epoch": 0.23820026466696073, "grad_norm": 7.53125, "learning_rate": 2.8099693217402807e-06, "loss": 1.545342254638672, "step": 135 }, { "epoch": 0.24702249669166298, "grad_norm": 2.125, "learning_rate": 2.6693805402077123e-06, "loss": 1.4941516876220704, "step": 140 }, { "epoch": 0.25584472871636527, "grad_norm": 2.828125, "learning_rate": 2.52825113118245e-06, "loss": 1.5371488571166991, "step": 145 }, { "epoch": 0.2646669607410675, "grad_norm": 2.328125, "learning_rate": 2.3870315504160995e-06, "loss": 1.4182353019714355, "step": 150 }, { "epoch": 0.2734891927657697, "grad_norm": 2.3125, "learning_rate": 2.24617254146973e-06, "loss": 1.524265193939209, "step": 155 }, { "epoch": 0.282311424790472, "grad_norm": 2.28125, "learning_rate": 2.1061236970340756e-06, "loss": 1.47442569732666, "step": 160 }, { "epoch": 0.2911336568151742, "grad_norm": 5.0, "learning_rate": 1.9673320239230783e-06, "loss": 1.4985308647155762, "step": 165 }, { "epoch": 0.2999558888398765, "grad_norm": 2.421875, "learning_rate": 1.830240516321008e-06, "loss": 1.491645050048828, "step": 170 }, { "epoch": 0.30877812086457873, "grad_norm": 4.125, "learning_rate": 1.6952867418370707e-06, "loss": 1.509144115447998, "step": 175 }, { "epoch": 0.317600352889281, "grad_norm": 4.625, "learning_rate": 1.562901444880508e-06, "loss": 1.464055347442627, "step": 180 }, { "epoch": 0.32642258491398324, "grad_norm": 6.375, "learning_rate": 1.4335071718139379e-06, "loss": 1.5107802391052245, "step": 185 }, { "epoch": 0.33524481693868546, "grad_norm": 2.8125, "learning_rate": 1.3075169222731573e-06, "loss": 1.52694730758667, "step": 190 }, { "epoch": 0.34406704896338774, "grad_norm": 2.09375, "learning_rate": 1.1853328309581139e-06, "loss": 1.4959157943725585, "step": 195 }, { "epoch": 0.35288928098808997, "grad_norm": 2.390625, "learning_rate": 1.0673448841024875e-06, "loss": 1.5017786979675294, "step": 200 }, { "epoch": 0.36171151301279225, "grad_norm": 2.625, "learning_rate": 9.53929674718668e-07, "loss": 1.5668895721435547, "step": 205 }, { "epoch": 0.3705337450374945, "grad_norm": 2.296875, "learning_rate": 8.454492005910942e-07, "loss": 1.5082152366638184, "step": 210 }, { "epoch": 0.37935597706219676, "grad_norm": 2.421875, "learning_rate": 7.422497088545436e-07, "loss": 1.5332984924316406, "step": 215 }, { "epoch": 0.388178209086899, "grad_norm": 3.265625, "learning_rate": 6.446605908452122e-07, "loss": 1.5301700592041017, "step": 220 }, { "epoch": 0.3970004411116012, "grad_norm": 2.203125, "learning_rate": 5.529933307520102e-07, "loss": 1.5099596977233887, "step": 225 }, { "epoch": 0.4058226731363035, "grad_norm": 2.25, "learning_rate": 4.6754051142374275e-07, "loss": 1.4918466567993165, "step": 230 }, { "epoch": 0.4146449051610057, "grad_norm": 5.0, "learning_rate": 3.8857488050544903e-07, "loss": 1.5015207290649415, "step": 235 }, { "epoch": 0.423467137185708, "grad_norm": 2.796875, "learning_rate": 3.163484798845862e-07, "loss": 1.5009549140930176, "step": 240 }, { "epoch": 0.4322893692104102, "grad_norm": 3.21875, "learning_rate": 2.5109184122568797e-07, "loss": 1.5147621154785156, "step": 245 }, { "epoch": 0.4411116012351125, "grad_norm": 1.9765625, "learning_rate": 1.9301325016119338e-07, "loss": 1.5966561317443848, "step": 250 }, { "epoch": 0.44993383325981473, "grad_norm": 2.609375, "learning_rate": 1.4229808148697732e-07, "loss": 1.5000021934509278, "step": 255 }, { "epoch": 0.45875606528451696, "grad_norm": 2.15625, "learning_rate": 9.91082074845215e-08, "loss": 1.4867940902709962, "step": 260 }, { "epoch": 0.46757829730921924, "grad_norm": 2.234375, "learning_rate": 6.358148125822e-08, "loss": 1.4553143501281738, "step": 265 }, { "epoch": 0.47640052933392146, "grad_norm": 5.40625, "learning_rate": 3.583129673691427e-08, "loss": 1.5283794403076172, "step": 270 }, { "epoch": 0.48522276135862374, "grad_norm": 4.625, "learning_rate": 1.5946226744029402e-08, "loss": 1.4848525047302246, "step": 275 }, { "epoch": 0.49404499338332597, "grad_norm": 2.75, "learning_rate": 3.989740291526212e-09, "loss": 1.5114261627197265, "step": 280 }, { "epoch": 0.5011027790030878, "step": 284, "total_flos": 1.3705181662943232e+17, "train_loss": 1.5080752087311007, "train_runtime": 2339.7979, "train_samples_per_second": 0.484, "train_steps_per_second": 0.121 } ], "logging_steps": 5, "max_steps": 284, "num_input_tokens_seen": 0, "num_train_epochs": 1, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": false, "should_training_stop": false }, "attributes": {} } }, "total_flos": 1.3705181662943232e+17, "train_batch_size": 1, "trial_name": null, "trial_params": null }