Files
v38_v29_r3_aggressive_lr7e7…/trainer_state.json

366 lines
8.8 KiB
JSON
Raw Permalink Normal View History

{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 1.0,
"eval_steps": 500,
"global_step": 461,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.021715526601520086,
"grad_norm": 5.0625,
"learning_rate": 4.5000000000000003e-07,
"loss": 2.309475326538086,
"step": 10
},
{
"epoch": 0.04343105320304017,
"grad_norm": 9.6875,
"learning_rate": 6.997839182620514e-07,
"loss": 2.2500890731811523,
"step": 20
},
{
"epoch": 0.06514657980456026,
"grad_norm": 7.90625,
"learning_rate": 6.980568648741917e-07,
"loss": 2.348200798034668,
"step": 30
},
{
"epoch": 0.08686210640608034,
"grad_norm": 5.34375,
"learning_rate": 6.946112854013265e-07,
"loss": 1.9902046203613282,
"step": 40
},
{
"epoch": 0.10857763300760044,
"grad_norm": 14.75,
"learning_rate": 6.894641923457209e-07,
"loss": 2.356439399719238,
"step": 50
},
{
"epoch": 0.13029315960912052,
"grad_norm": 7.25,
"learning_rate": 6.826409994100501e-07,
"loss": 2.275517463684082,
"step": 60
},
{
"epoch": 0.15200868621064062,
"grad_norm": 7.9375,
"learning_rate": 6.74175396017585e-07,
"loss": 1.863471221923828,
"step": 70
},
{
"epoch": 0.1737242128121607,
"grad_norm": 13.125,
"learning_rate": 6.641091809711172e-07,
"loss": 2.199751091003418,
"step": 80
},
{
"epoch": 0.19543973941368079,
"grad_norm": 13.75,
"learning_rate": 6.52492056071934e-07,
"loss": 2.2227617263793946,
"step": 90
},
{
"epoch": 0.21715526601520088,
"grad_norm": 7.875,
"learning_rate": 6.393813807178434e-07,
"loss": 2.2083740234375,
"step": 100
},
{
"epoch": 0.23887079261672095,
"grad_norm": 16.625,
"learning_rate": 6.248418886919186e-07,
"loss": 2.2678974151611326,
"step": 110
},
{
"epoch": 0.26058631921824105,
"grad_norm": 11.6875,
"learning_rate": 6.089453685403166e-07,
"loss": 2.0732358932495116,
"step": 120
},
{
"epoch": 0.28230184581976114,
"grad_norm": 12.9375,
"learning_rate": 5.917703091172979e-07,
"loss": 2.0399072647094725,
"step": 130
},
{
"epoch": 0.30401737242128124,
"grad_norm": 9.3125,
"learning_rate": 5.73401512047564e-07,
"loss": 2.0727848052978515,
"step": 140
},
{
"epoch": 0.3257328990228013,
"grad_norm": 24.25,
"learning_rate": 5.539296730193784e-07,
"loss": 2.0733610153198243,
"step": 150
},
{
"epoch": 0.3474484256243214,
"grad_norm": 5.0625,
"learning_rate": 5.334509339758256e-07,
"loss": 2.1352399826049804,
"step": 160
},
{
"epoch": 0.3691639522258415,
"grad_norm": 7.5,
"learning_rate": 5.120664084152621e-07,
"loss": 2.003406524658203,
"step": 170
},
{
"epoch": 0.39087947882736157,
"grad_norm": 5.875,
"learning_rate": 4.898816821447787e-07,
"loss": 2.21402530670166,
"step": 180
},
{
"epoch": 0.41259500542888167,
"grad_norm": 12.1875,
"learning_rate": 4.6700629195170006e-07,
"loss": 2.0832319259643555,
"step": 190
},
{
"epoch": 0.43431053203040176,
"grad_norm": 11.4375,
"learning_rate": 4.43553184767171e-07,
"loss": 2.305379295349121,
"step": 200
},
{
"epoch": 0.4560260586319218,
"grad_norm": 7.71875,
"learning_rate": 4.1963815999220493e-07,
"loss": 2.1995058059692383,
"step": 210
},
{
"epoch": 0.4777415852334419,
"grad_norm": 9.8125,
"learning_rate": 3.953792977397001e-07,
"loss": 2.07507266998291,
"step": 220
},
{
"epoch": 0.499457111834962,
"grad_norm": 8.625,
"learning_rate": 3.708963758154715e-07,
"loss": 1.987565040588379,
"step": 230
},
{
"epoch": 0.5211726384364821,
"grad_norm": 17.125,
"learning_rate": 3.463102783169471e-07,
"loss": 1.974489974975586,
"step": 240
},
{
"epoch": 0.5428881650380022,
"grad_norm": 5.90625,
"learning_rate": 3.217423987695643e-07,
"loss": 2.0915807723999023,
"step": 250
},
{
"epoch": 0.5646036916395223,
"grad_norm": 7.78125,
"learning_rate": 2.9731404074787175e-07,
"loss": 2.112474060058594,
"step": 260
},
{
"epoch": 0.5863192182410424,
"grad_norm": 21.625,
"learning_rate": 2.73145818940763e-07,
"loss": 2.2007789611816406,
"step": 270
},
{
"epoch": 0.6080347448425625,
"grad_norm": 13.8125,
"learning_rate": 2.4935706361807427e-07,
"loss": 2.275239181518555,
"step": 280
},
{
"epoch": 0.6297502714440825,
"grad_norm": 8.6875,
"learning_rate": 2.2606523143898274e-07,
"loss": 2.0178400039672852,
"step": 290
},
{
"epoch": 0.6514657980456026,
"grad_norm": 6.21875,
"learning_rate": 2.033853255113336e-07,
"loss": 2.1944948196411134,
"step": 300
},
{
"epoch": 0.6731813246471227,
"grad_norm": 18.75,
"learning_rate": 1.8142932756534272e-07,
"loss": 2.111407661437988,
"step": 310
},
{
"epoch": 0.6948968512486428,
"grad_norm": 14.875,
"learning_rate": 1.603056450453112e-07,
"loss": 2.0520841598510744,
"step": 320
},
{
"epoch": 0.7166123778501629,
"grad_norm": 4.0,
"learning_rate": 1.401185758493289e-07,
"loss": 1.9921758651733399,
"step": 330
},
{
"epoch": 0.738327904451683,
"grad_norm": 10.125,
"learning_rate": 1.2096779335980598e-07,
"loss": 2.2591268539428713,
"step": 340
},
{
"epoch": 0.760043431053203,
"grad_norm": 18.375,
"learning_rate": 1.0294785430748916e-07,
"loss": 2.3327383041381835,
"step": 350
},
{
"epoch": 0.7817589576547231,
"grad_norm": 6.65625,
"learning_rate": 8.6147731898877e-08,
"loss": 2.176487350463867,
"step": 360
},
{
"epoch": 0.8034744842562432,
"grad_norm": 7.03125,
"learning_rate": 7.065037651220986e-08,
"loss": 2.1322710037231447,
"step": 370
},
{
"epoch": 0.8251900108577633,
"grad_norm": 10.0625,
"learning_rate": 5.653230613109414e-08,
"loss": 2.1231149673461913,
"step": 380
},
{
"epoch": 0.8469055374592834,
"grad_norm": 4.96875,
"learning_rate": 4.38632285379887e-08,
"loss": 2.1640350341796877,
"step": 390
},
{
"epoch": 0.8686210640608035,
"grad_norm": 7.3125,
"learning_rate": 3.27056971329667e-08,
"loss": 2.0533117294311523,
"step": 400
},
{
"epoch": 0.8903365906623235,
"grad_norm": 8.5625,
"learning_rate": 2.3114802077145063e-08,
"loss": 2.199868011474609,
"step": 410
},
{
"epoch": 0.9120521172638436,
"grad_norm": 7.5625,
"learning_rate": 1.5137898285755707e-08,
"loss": 2.2271245956420898,
"step": 420
},
{
"epoch": 0.9337676438653637,
"grad_norm": 5.71875,
"learning_rate": 8.814371613888839e-09,
"loss": 2.0086801528930662,
"step": 430
},
{
"epoch": 0.9554831704668838,
"grad_norm": 5.21875,
"learning_rate": 4.175444389363636e-09,
"loss": 1.9937088012695312,
"step": 440
},
{
"epoch": 0.9771986970684039,
"grad_norm": 12.0625,
"learning_rate": 1.2440212529050531e-09,
"loss": 2.1526458740234373,
"step": 450
},
{
"epoch": 0.998914223669924,
"grad_norm": 8.75,
"learning_rate": 3.4576066788472026e-11,
"loss": 2.2378005981445312,
"step": 460
},
{
"epoch": 1.0,
"step": 461,
"total_flos": 2.197680639479255e+17,
"train_loss": 2.143569690286467,
"train_runtime": 3558.7051,
"train_samples_per_second": 0.518,
"train_steps_per_second": 0.13
}
],
"logging_steps": 10,
"max_steps": 461,
"num_input_tokens_seen": 0,
"num_train_epochs": 1,
"save_steps": 500,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": false,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 2.197680639479255e+17,
"train_batch_size": 1,
"trial_name": null,
"trial_params": null
}