Files
v35_v29_cot_r3_balanced_lr6…/trainer_state.json

380 lines
9.1 KiB
JSON
Raw Normal View History

{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 1.0,
"eval_steps": 500,
"global_step": 485,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.020650490449148167,
"grad_norm": 4.40625,
"learning_rate": 3.6e-07,
"loss": 2.075181579589844,
"step": 10
},
{
"epoch": 0.041300980898296334,
"grad_norm": 16.0,
"learning_rate": 5.998927766432351e-07,
"loss": 2.085246467590332,
"step": 20
},
{
"epoch": 0.061951471347444505,
"grad_norm": 30.0,
"learning_rate": 5.986873939340487e-07,
"loss": 2.2069232940673826,
"step": 30
},
{
"epoch": 0.08260196179659267,
"grad_norm": 5.53125,
"learning_rate": 5.961480008160849e-07,
"loss": 2.1064844131469727,
"step": 40
},
{
"epoch": 0.10325245224574084,
"grad_norm": 7.53125,
"learning_rate": 5.922859388354552e-07,
"loss": 2.148875617980957,
"step": 50
},
{
"epoch": 0.12390294269488901,
"grad_norm": 22.25,
"learning_rate": 5.871184568984952e-07,
"loss": 2.2333826065063476,
"step": 60
},
{
"epoch": 0.14455343314403718,
"grad_norm": 25.0,
"learning_rate": 5.806686342339606e-07,
"loss": 2.1595861434936525,
"step": 70
},
{
"epoch": 0.16520392359318534,
"grad_norm": 10.8125,
"learning_rate": 5.729652773155893e-07,
"loss": 2.0489675521850588,
"step": 80
},
{
"epoch": 0.1858544140423335,
"grad_norm": 6.375,
"learning_rate": 5.640427912053975e-07,
"loss": 2.1140844345092775,
"step": 90
},
{
"epoch": 0.20650490449148168,
"grad_norm": 14.6875,
"learning_rate": 5.539410258923193e-07,
"loss": 2.1626249313354493,
"step": 100
},
{
"epoch": 0.22715539494062983,
"grad_norm": 6.59375,
"learning_rate": 5.427050983124842e-07,
"loss": 1.9297166824340821,
"step": 110
},
{
"epoch": 0.24780588538977802,
"grad_norm": 23.875,
"learning_rate": 5.303851908460263e-07,
"loss": 2.105946350097656,
"step": 120
},
{
"epoch": 0.2684563758389262,
"grad_norm": 16.375,
"learning_rate": 5.170363271903941e-07,
"loss": 2.3158504486083986,
"step": 130
},
{
"epoch": 0.28910686628807436,
"grad_norm": 9.8125,
"learning_rate": 5.027181266111595e-07,
"loss": 2.025718116760254,
"step": 140
},
{
"epoch": 0.3097573567372225,
"grad_norm": 12.3125,
"learning_rate": 4.874945376679055e-07,
"loss": 2.0087594985961914,
"step": 150
},
{
"epoch": 0.3304078471863707,
"grad_norm": 8.8125,
"learning_rate": 4.71433552604435e-07,
"loss": 1.9722934722900392,
"step": 160
},
{
"epoch": 0.35105833763551886,
"grad_norm": 19.875,
"learning_rate": 4.5460690367890496e-07,
"loss": 2.084832954406738,
"step": 170
},
{
"epoch": 0.371708828084667,
"grad_norm": 16.75,
"learning_rate": 4.37089742790148e-07,
"loss": 1.9973403930664062,
"step": 180
},
{
"epoch": 0.39235931853381517,
"grad_norm": 8.125,
"learning_rate": 4.1896030583104884e-07,
"loss": 1.9446184158325195,
"step": 190
},
{
"epoch": 0.41300980898296336,
"grad_norm": 20.625,
"learning_rate": 4.0029956326805454e-07,
"loss": 2.216297721862793,
"step": 200
},
{
"epoch": 0.43366029943211154,
"grad_norm": 8.0625,
"learning_rate": 3.8119085850741565e-07,
"loss": 1.9691858291625977,
"step": 210
},
{
"epoch": 0.45431078988125967,
"grad_norm": 5.78125,
"learning_rate": 3.617195356633018e-07,
"loss": 2.18753604888916,
"step": 220
},
{
"epoch": 0.47496128033040785,
"grad_norm": 4.96875,
"learning_rate": 3.419725583902706e-07,
"loss": 2.077012825012207,
"step": 230
},
{
"epoch": 0.49561177077955604,
"grad_norm": 12.6875,
"learning_rate": 3.2203812148247463e-07,
"loss": 2.1807201385498045,
"step": 240
},
{
"epoch": 0.5162622612287042,
"grad_norm": 14.25,
"learning_rate": 3.020052569743027e-07,
"loss": 2.1047475814819334,
"step": 250
},
{
"epoch": 0.5369127516778524,
"grad_norm": 10.4375,
"learning_rate": 2.819634365016999e-07,
"loss": 1.9689088821411134,
"step": 260
},
{
"epoch": 0.5575632421270005,
"grad_norm": 5.125,
"learning_rate": 2.6200217170012295e-07,
"loss": 2.0879981994628904,
"step": 270
},
{
"epoch": 0.5782137325761487,
"grad_norm": 11.75,
"learning_rate": 2.422106144238464e-07,
"loss": 2.080374336242676,
"step": 280
},
{
"epoch": 0.5988642230252968,
"grad_norm": 6.4375,
"learning_rate": 2.2267715857213983e-07,
"loss": 2.179820251464844,
"step": 290
},
{
"epoch": 0.619514713474445,
"grad_norm": 8.0,
"learning_rate": 2.0348904530065538e-07,
"loss": 1.9253725051879882,
"step": 300
},
{
"epoch": 0.6401652039235932,
"grad_norm": 6.625,
"learning_rate": 1.8473197338124887e-07,
"loss": 1.8352821350097657,
"step": 310
},
{
"epoch": 0.6608156943727413,
"grad_norm": 6.0625,
"learning_rate": 1.66489716450461e-07,
"loss": 2.013896369934082,
"step": 320
},
{
"epoch": 0.6814661848218895,
"grad_norm": 5.4375,
"learning_rate": 1.4884374885611996e-07,
"loss": 2.13570613861084,
"step": 330
},
{
"epoch": 0.7021166752710377,
"grad_norm": 9.125,
"learning_rate": 1.3187288177312556e-07,
"loss": 2.071014404296875,
"step": 340
},
{
"epoch": 0.7227671657201858,
"grad_norm": 13.375,
"learning_rate": 1.1565291121361143e-07,
"loss": 2.260077476501465,
"step": 350
},
{
"epoch": 0.743417656169334,
"grad_norm": 13.5,
"learning_rate": 1.0025627950355415e-07,
"loss": 2.126547431945801,
"step": 360
},
{
"epoch": 0.7640681466184822,
"grad_norm": 12.4375,
"learning_rate": 8.575175173775983e-08,
"loss": 2.061558151245117,
"step": 370
},
{
"epoch": 0.7847186370676303,
"grad_norm": 9.625,
"learning_rate": 7.220410865825828e-08,
"loss": 2.0802581787109373,
"step": 380
},
{
"epoch": 0.8053691275167785,
"grad_norm": 10.6875,
"learning_rate": 5.967385732778304e-08,
"loss": 1.9892038345336913,
"step": 390
},
{
"epoch": 0.8260196179659267,
"grad_norm": 8.75,
"learning_rate": 4.821696089054069e-08,
"loss": 2.121898651123047,
"step": 400
},
{
"epoch": 0.8466701084150748,
"grad_norm": 4.53125,
"learning_rate": 3.7884588627223946e-08,
"loss": 1.8785961151123047,
"step": 410
},
{
"epoch": 0.8673205988642231,
"grad_norm": 6.28125,
"learning_rate": 2.8722887420582297e-08,
"loss": 2.034098815917969,
"step": 420
},
{
"epoch": 0.8879710893133712,
"grad_norm": 17.0,
"learning_rate": 2.077277565224146e-08,
"loss": 2.088451385498047,
"step": 430
},
{
"epoch": 0.9086215797625193,
"grad_norm": 17.5,
"learning_rate": 1.4069760451277756e-08,
"loss": 2.0510417938232424,
"step": 440
},
{
"epoch": 0.9292720702116676,
"grad_norm": 18.25,
"learning_rate": 8.64377911076044e-09,
"loss": 2.012544631958008,
"step": 450
},
{
"epoch": 0.9499225606608157,
"grad_norm": 10.0,
"learning_rate": 4.519065380533282e-09,
"loss": 2.140595817565918,
"step": 460
},
{
"epoch": 0.9705730511099638,
"grad_norm": 11.625,
"learning_rate": 1.7140412334049615e-09,
"loss": 1.9902917861938476,
"step": 470
},
{
"epoch": 0.9912235415591121,
"grad_norm": 10.625,
"learning_rate": 2.4123458814648835e-10,
"loss": 2.080463409423828,
"step": 480
},
{
"epoch": 1.0,
"step": 485,
"total_flos": 2.3199594469350605e+17,
"train_loss": 2.076033425085323,
"train_runtime": 3739.4992,
"train_samples_per_second": 0.518,
"train_steps_per_second": 0.13
}
],
"logging_steps": 10,
"max_steps": 485,
"num_input_tokens_seen": 0,
"num_train_epochs": 1,
"save_steps": 500,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": false,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 2.3199594469350605e+17,
"train_batch_size": 1,
"trial_name": null,
"trial_params": null
}