{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 1.0417482061317678, "eval_steps": 500, "global_step": 200, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.02609262883235486, "grad_norm": 90.7743148803711, "learning_rate": 5.128205128205128e-07, "loss": 3.4203, "step": 5 }, { "epoch": 0.05218525766470972, "grad_norm": 52.55401611328125, "learning_rate": 1.153846153846154e-06, "loss": 2.8478, "step": 10 }, { "epoch": 0.07827788649706457, "grad_norm": 17.9848575592041, "learning_rate": 1.794871794871795e-06, "loss": 1.524, "step": 15 }, { "epoch": 0.10437051532941943, "grad_norm": 9.579212188720703, "learning_rate": 2.435897435897436e-06, "loss": 0.8379, "step": 20 }, { "epoch": 0.1304631441617743, "grad_norm": 6.430282115936279, "learning_rate": 3.0769230769230774e-06, "loss": 0.4496, "step": 25 }, { "epoch": 0.15655577299412915, "grad_norm": 5.243109703063965, "learning_rate": 3.7179487179487184e-06, "loss": 0.2512, "step": 30 }, { "epoch": 0.182648401826484, "grad_norm": 5.690948963165283, "learning_rate": 4.358974358974359e-06, "loss": 0.183, "step": 35 }, { "epoch": 0.20874103065883887, "grad_norm": 4.116969585418701, "learning_rate": 5e-06, "loss": 0.1788, "step": 40 }, { "epoch": 0.23483365949119372, "grad_norm": 2.8701882362365723, "learning_rate": 4.9974091841168195e-06, "loss": 0.1605, "step": 45 }, { "epoch": 0.2609262883235486, "grad_norm": 7.059335231781006, "learning_rate": 4.989642106328829e-06, "loss": 0.1394, "step": 50 }, { "epoch": 0.28701891715590344, "grad_norm": 3.2651431560516357, "learning_rate": 4.976714865090827e-06, "loss": 0.083, "step": 55 }, { "epoch": 0.3131115459882583, "grad_norm": 2.457581043243408, "learning_rate": 4.958654254084356e-06, "loss": 0.0896, "step": 60 }, { "epoch": 0.33920417482061316, "grad_norm": 2.72770619392395, "learning_rate": 4.935497706683698e-06, "loss": 0.0991, "step": 65 }, { "epoch": 0.365296803652968, "grad_norm": 2.0538723468780518, "learning_rate": 4.907293218369499e-06, "loss": 0.0979, "step": 70 }, { "epoch": 0.3913894324853229, "grad_norm": 2.1862363815307617, "learning_rate": 4.874099247250799e-06, "loss": 0.0933, "step": 75 }, { "epoch": 0.41748206131767773, "grad_norm": 2.605069637298584, "learning_rate": 4.835984592901678e-06, "loss": 0.0959, "step": 80 }, { "epoch": 0.4435746901500326, "grad_norm": 1.2768350839614868, "learning_rate": 4.793028253763633e-06, "loss": 0.0749, "step": 85 }, { "epoch": 0.46966731898238745, "grad_norm": 4.374709606170654, "learning_rate": 4.745319263409241e-06, "loss": 0.0991, "step": 90 }, { "epoch": 0.4957599478147423, "grad_norm": 1.9398008584976196, "learning_rate": 4.692956506006486e-06, "loss": 0.0833, "step": 95 }, { "epoch": 0.5218525766470972, "grad_norm": 1.7450673580169678, "learning_rate": 4.636048511366222e-06, "loss": 0.0804, "step": 100 }, { "epoch": 0.547945205479452, "grad_norm": 1.712754249572754, "learning_rate": 4.5747132299975634e-06, "loss": 0.0867, "step": 105 }, { "epoch": 0.5740378343118069, "grad_norm": 1.083371877670288, "learning_rate": 4.509077788637446e-06, "loss": 0.0688, "step": 110 }, { "epoch": 0.6001304631441617, "grad_norm": 1.936558485031128, "learning_rate": 4.43927822676105e-06, "loss": 0.0772, "step": 115 }, { "epoch": 0.6262230919765166, "grad_norm": 1.4485265016555786, "learning_rate": 4.3654592146192146e-06, "loss": 0.0767, "step": 120 }, { "epoch": 0.6523157208088715, "grad_norm": 1.4740047454833984, "learning_rate": 4.287773753387249e-06, "loss": 0.0715, "step": 125 }, { "epoch": 0.6784083496412263, "grad_norm": 1.7988250255584717, "learning_rate": 4.206382858046636e-06, "loss": 0.0823, "step": 130 }, { "epoch": 0.7045009784735812, "grad_norm": 0.9675837159156799, "learning_rate": 4.12145522365689e-06, "loss": 0.0955, "step": 135 }, { "epoch": 0.730593607305936, "grad_norm": 1.3645009994506836, "learning_rate": 4.033166875709291e-06, "loss": 0.062, "step": 140 }, { "epoch": 0.7566862361382909, "grad_norm": 1.3581523895263672, "learning_rate": 3.941700805287169e-06, "loss": 0.0868, "step": 145 }, { "epoch": 0.7827788649706457, "grad_norm": 1.8347728252410889, "learning_rate": 3.84724658978894e-06, "loss": 0.0681, "step": 150 }, { "epoch": 0.8088714938030006, "grad_norm": 1.278552770614624, "learning_rate": 3.7500000000000005e-06, "loss": 0.0709, "step": 155 }, { "epoch": 0.8349641226353555, "grad_norm": 1.676578402519226, "learning_rate": 3.650162594327881e-06, "loss": 0.0714, "step": 160 }, { "epoch": 0.8610567514677103, "grad_norm": 0.9458003640174866, "learning_rate": 3.5479413010416606e-06, "loss": 0.081, "step": 165 }, { "epoch": 0.8871493803000652, "grad_norm": 1.0945783853530884, "learning_rate": 3.443547989381536e-06, "loss": 0.0715, "step": 170 }, { "epoch": 0.91324200913242, "grad_norm": 1.2435601949691772, "learning_rate": 3.3371990304274654e-06, "loss": 0.0638, "step": 175 }, { "epoch": 0.9393346379647749, "grad_norm": 1.2059943675994873, "learning_rate": 3.2291148486370626e-06, "loss": 0.0683, "step": 180 }, { "epoch": 0.9654272667971298, "grad_norm": 0.6467380523681641, "learning_rate": 3.11951946498225e-06, "loss": 0.0709, "step": 185 }, { "epoch": 0.9915198956294846, "grad_norm": 0.8123487830162048, "learning_rate": 3.0086400326315853e-06, "loss": 0.0601, "step": 190 }, { "epoch": 1.015655577299413, "grad_norm": 1.2716907262802124, "learning_rate": 2.896706366140629e-06, "loss": 0.0579, "step": 195 }, { "epoch": 1.0417482061317678, "grad_norm": 1.656204104423523, "learning_rate": 2.7839504651261873e-06, "loss": 0.0623, "step": 200 } ], "logging_steps": 5, "max_steps": 384, "num_input_tokens_seen": 0, "num_train_epochs": 2, "save_steps": 200, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 4.890173722188964e+17, "train_batch_size": 4, "trial_name": null, "trial_params": null }