{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 1.0, "eval_steps": 500, "global_step": 326, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.03067484662576687, "grad_norm": 8.0, "learning_rate": 1.8e-05, "loss": 2.659077453613281, "step": 10 }, { "epoch": 0.06134969325153374, "grad_norm": 3.296875, "learning_rate": 1.995999715857997e-05, "loss": 2.113694763183594, "step": 20 }, { "epoch": 0.09202453987730061, "grad_norm": 2.984375, "learning_rate": 1.9822126571413616e-05, "loss": 1.8163612365722657, "step": 30 }, { "epoch": 0.12269938650306748, "grad_norm": 3.375, "learning_rate": 1.9587255619128648e-05, "loss": 1.7940385818481446, "step": 40 }, { "epoch": 0.15337423312883436, "grad_norm": 2.90625, "learning_rate": 1.9257703816543144e-05, "loss": 1.702846145629883, "step": 50 }, { "epoch": 0.18404907975460122, "grad_norm": 2.890625, "learning_rate": 1.8836725718049562e-05, "loss": 1.6503068923950195, "step": 60 }, { "epoch": 0.2147239263803681, "grad_norm": 3.265625, "learning_rate": 1.8328478776615336e-05, "loss": 1.6176418304443358, "step": 70 }, { "epoch": 0.24539877300613497, "grad_norm": 3.125, "learning_rate": 1.7737982286028938e-05, "loss": 1.649836540222168, "step": 80 }, { "epoch": 0.27607361963190186, "grad_norm": 2.671875, "learning_rate": 1.7071067811865477e-05, "loss": 1.6427200317382813, "step": 90 }, { "epoch": 0.3067484662576687, "grad_norm": 2.5625, "learning_rate": 1.6334321600700612e-05, "loss": 1.5673911094665527, "step": 100 }, { "epoch": 0.3374233128834356, "grad_norm": 3.453125, "learning_rate": 1.5535019536322158e-05, "loss": 1.5665663719177245, "step": 110 }, { "epoch": 0.36809815950920244, "grad_norm": 3.28125, "learning_rate": 1.4681055285292138e-05, "loss": 1.549239158630371, "step": 120 }, { "epoch": 0.3987730061349693, "grad_norm": 3.015625, "learning_rate": 1.3780862341472183e-05, "loss": 1.5795641899108888, "step": 130 }, { "epoch": 0.4294478527607362, "grad_norm": 2.90625, "learning_rate": 1.2843330739377003e-05, "loss": 1.5186445236206054, "step": 140 }, { "epoch": 0.4601226993865031, "grad_norm": 2.890625, "learning_rate": 1.1877719258869827e-05, "loss": 1.5579788208007812, "step": 150 }, { "epoch": 0.49079754601226994, "grad_norm": 3.578125, "learning_rate": 1.0893563988239773e-05, "loss": 1.5645319938659668, "step": 160 }, { "epoch": 0.5214723926380368, "grad_norm": 2.671875, "learning_rate": 9.900584148664705e-06, "loss": 1.5919049263000489, "step": 170 }, { "epoch": 0.5521472392638037, "grad_norm": 2.25, "learning_rate": 8.908586110108794e-06, "loss": 1.571799087524414, "step": 180 }, { "epoch": 0.5828220858895705, "grad_norm": 2.625, "learning_rate": 7.927366546564911e-06, "loss": 1.5389130592346192, "step": 190 }, { "epoch": 0.6134969325153374, "grad_norm": 2.640625, "learning_rate": 6.966615687051517e-06, "loss": 1.5266441345214843, "step": 200 }, { "epoch": 0.6441717791411042, "grad_norm": 3.015625, "learning_rate": 6.03582161782806e-06, "loss": 1.5582975387573241, "step": 210 }, { "epoch": 0.6748466257668712, "grad_norm": 2.921875, "learning_rate": 5.144176580911431e-06, "loss": 1.5671936988830566, "step": 220 }, { "epoch": 0.7055214723926381, "grad_norm": 2.4375, "learning_rate": 4.3004861942610575e-06, "loss": 1.5731188774108886, "step": 230 }, { "epoch": 0.7361963190184049, "grad_norm": 3.328125, "learning_rate": 3.513082490146864e-06, "loss": 1.5142654418945312, "step": 240 }, { "epoch": 0.7668711656441718, "grad_norm": 3.046875, "learning_rate": 2.7897416305068325e-06, "loss": 1.5008227348327636, "step": 250 }, { "epoch": 0.7975460122699386, "grad_norm": 2.875, "learning_rate": 2.137607111912734e-06, "loss": 1.585028839111328, "step": 260 }, { "epoch": 0.8282208588957055, "grad_norm": 2.640625, "learning_rate": 1.5631192185484557e-06, "loss": 1.571006679534912, "step": 270 }, { "epoch": 0.8588957055214724, "grad_norm": 3.046875, "learning_rate": 1.0719514199022473e-06, "loss": 1.5311025619506835, "step": 280 }, { "epoch": 0.8895705521472392, "grad_norm": 2.78125, "learning_rate": 6.689543412899913e-07, "loss": 1.5679447174072265, "step": 290 }, { "epoch": 0.9202453987730062, "grad_norm": 2.6875, "learning_rate": 3.5810786053987025e-07, "loss": 1.5297722816467285, "step": 300 }, { "epoch": 0.950920245398773, "grad_norm": 2.75, "learning_rate": 1.4248180391703614e-07, "loss": 1.5922201156616211, "step": 310 }, { "epoch": 0.9815950920245399, "grad_norm": 2.796875, "learning_rate": 2.420562944358329e-08, "loss": 1.512490463256836, "step": 320 }, { "epoch": 1.0, "step": 326, "total_flos": 1.563295846805975e+17, "train_loss": 1.6355595296145948, "train_runtime": 2594.717, "train_samples_per_second": 0.503, "train_steps_per_second": 0.126 } ], "logging_steps": 10, "max_steps": 326, "num_input_tokens_seen": 0, "num_train_epochs": 1, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": false, "should_training_stop": false }, "attributes": {} } }, "total_flos": 1.563295846805975e+17, "train_batch_size": 1, "trial_name": null, "trial_params": null }