Files
v34_v29_paper_mix_lr5e7_ep10/trainer_state.json

373 lines
8.9 KiB
JSON
Raw Normal View History

{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 1.0,
"eval_steps": 500,
"global_step": 476,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.021019442984760904,
"grad_norm": 5.53125,
"learning_rate": 3e-07,
"loss": 1.8080066680908202,
"step": 10
},
{
"epoch": 0.04203888596952181,
"grad_norm": 5.15625,
"learning_rate": 4.99907124535624e-07,
"loss": 1.9573022842407226,
"step": 20
},
{
"epoch": 0.0630583289542827,
"grad_norm": 8.125,
"learning_rate": 4.988630678996146e-07,
"loss": 1.93359317779541,
"step": 30
},
{
"epoch": 0.08407777193904362,
"grad_norm": 8.75,
"learning_rate": 4.966637232576035e-07,
"loss": 1.8890295028686523,
"step": 40
},
{
"epoch": 0.10509721492380451,
"grad_norm": 7.375,
"learning_rate": 4.933193005475595e-07,
"loss": 1.8508934020996093,
"step": 50
},
{
"epoch": 0.1261166579085654,
"grad_norm": 5.21875,
"learning_rate": 4.888453254618907e-07,
"loss": 1.8927783966064453,
"step": 60
},
{
"epoch": 0.14713610089332632,
"grad_norm": 5.125,
"learning_rate": 4.832625673730847e-07,
"loss": 1.8637098312377929,
"step": 70
},
{
"epoch": 0.16815554387808723,
"grad_norm": 11.0625,
"learning_rate": 4.765969429168031e-07,
"loss": 1.983097457885742,
"step": 80
},
{
"epoch": 0.18917498686284814,
"grad_norm": 5.21875,
"learning_rate": 4.6887939568002264e-07,
"loss": 1.919074821472168,
"step": 90
},
{
"epoch": 0.21019442984760903,
"grad_norm": 5.9375,
"learning_rate": 4.6014575255274225e-07,
"loss": 1.8743148803710938,
"step": 100
},
{
"epoch": 0.23121387283236994,
"grad_norm": 9.4375,
"learning_rate": 4.504365574101089e-07,
"loss": 1.9480220794677734,
"step": 110
},
{
"epoch": 0.2522333158171308,
"grad_norm": 6.1875,
"learning_rate": 4.3979688289705503e-07,
"loss": 1.8002925872802735,
"step": 120
},
{
"epoch": 0.27325275880189176,
"grad_norm": 7.125,
"learning_rate": 4.282761211891912e-07,
"loss": 1.915324592590332,
"step": 130
},
{
"epoch": 0.29427220178665264,
"grad_norm": 12.625,
"learning_rate": 4.1592775470129854e-07,
"loss": 1.985239601135254,
"step": 140
},
{
"epoch": 0.3152916447714136,
"grad_norm": 7.28125,
"learning_rate": 4.028091078078516e-07,
"loss": 1.8037527084350586,
"step": 150
},
{
"epoch": 0.33631108775617446,
"grad_norm": 6.46875,
"learning_rate": 3.8898108072815116e-07,
"loss": 1.8475725173950195,
"step": 160
},
{
"epoch": 0.35733053074093535,
"grad_norm": 9.375,
"learning_rate": 3.7450786681144234e-07,
"loss": 1.8196781158447266,
"step": 170
},
{
"epoch": 0.3783499737256963,
"grad_norm": 6.90625,
"learning_rate": 3.5945665453445397e-07,
"loss": 1.813003921508789,
"step": 180
},
{
"epoch": 0.39936941671045717,
"grad_norm": 5.8125,
"learning_rate": 3.4389731559476716e-07,
"loss": 1.8163671493530273,
"step": 190
},
{
"epoch": 0.42038885969521805,
"grad_norm": 9.5,
"learning_rate": 3.279020805479643e-07,
"loss": 1.853123092651367,
"step": 200
},
{
"epoch": 0.441408302679979,
"grad_norm": 6.59375,
"learning_rate": 3.1154520349433735e-07,
"loss": 1.8787429809570313,
"step": 210
},
{
"epoch": 0.4624277456647399,
"grad_norm": 5.875,
"learning_rate": 2.9490261737176715e-07,
"loss": 1.9475088119506836,
"step": 220
},
{
"epoch": 0.4834471886495008,
"grad_norm": 12.75,
"learning_rate": 2.780515814549969e-07,
"loss": 1.9444992065429687,
"step": 230
},
{
"epoch": 0.5044666316342616,
"grad_norm": 4.96875,
"learning_rate": 2.6107032269769897e-07,
"loss": 1.7953880310058594,
"step": 240
},
{
"epoch": 0.5254860746190226,
"grad_norm": 7.53125,
"learning_rate": 2.440376725823213e-07,
"loss": 1.9037723541259766,
"step": 250
},
{
"epoch": 0.5465055176037835,
"grad_norm": 9.5625,
"learning_rate": 2.2703270116355234e-07,
"loss": 1.9526327133178711,
"step": 260
},
{
"epoch": 0.5675249605885444,
"grad_norm": 8.875,
"learning_rate": 2.101343500042697e-07,
"loss": 1.9550243377685548,
"step": 270
},
{
"epoch": 0.5885444035733053,
"grad_norm": 11.75,
"learning_rate": 1.934210657079821e-07,
"loss": 1.896200180053711,
"step": 280
},
{
"epoch": 0.6095638465580662,
"grad_norm": 4.53125,
"learning_rate": 1.7697043574900388e-07,
"loss": 1.9687370300292968,
"step": 290
},
{
"epoch": 0.6305832895428272,
"grad_norm": 4.90625,
"learning_rate": 1.608588282909325e-07,
"loss": 1.7474214553833007,
"step": 300
},
{
"epoch": 0.651602732527588,
"grad_norm": 12.1875,
"learning_rate": 1.4516103766548969e-07,
"loss": 1.952895164489746,
"step": 310
},
{
"epoch": 0.6726221755123489,
"grad_norm": 7.875,
"learning_rate": 1.2994993715750356e-07,
"loss": 2.002369689941406,
"step": 320
},
{
"epoch": 0.6936416184971098,
"grad_norm": 4.75,
"learning_rate": 1.1529614070789883e-07,
"loss": 1.9351655960083007,
"step": 330
},
{
"epoch": 0.7146610614818707,
"grad_norm": 5.9375,
"learning_rate": 1.0126767510515585e-07,
"loss": 1.9994436264038087,
"step": 340
},
{
"epoch": 0.7356805044666316,
"grad_norm": 6.875,
"learning_rate": 8.792966418701511e-08,
"loss": 2.0275642395019533,
"step": 350
},
{
"epoch": 0.7566999474513926,
"grad_norm": 5.28125,
"learning_rate": 7.534402651844351e-08,
"loss": 1.9153682708740234,
"step": 360
},
{
"epoch": 0.7777193904361535,
"grad_norm": 4.71875,
"learning_rate": 6.356918794932362e-08,
"loss": 1.9506032943725586,
"step": 370
},
{
"epoch": 0.7987388334209143,
"grad_norm": 22.125,
"learning_rate": 5.265981038624656e-08,
"loss": 1.9164735794067382,
"step": 380
},
{
"epoch": 0.8197582764056752,
"grad_norm": 10.0625,
"learning_rate": 4.266653803752299e-08,
"loss": 1.793134880065918,
"step": 390
},
{
"epoch": 0.8407777193904361,
"grad_norm": 5.6875,
"learning_rate": 3.363576230940962e-08,
"loss": 1.9492578506469727,
"step": 400
},
{
"epoch": 0.8617971623751971,
"grad_norm": 4.96875,
"learning_rate": 2.560940644496379e-08,
"loss": 1.8826881408691407,
"step": 410
},
{
"epoch": 0.882816605359958,
"grad_norm": 18.875,
"learning_rate": 1.8624730905291208e-08,
"loss": 1.80709228515625,
"step": 420
},
{
"epoch": 0.9038360483447189,
"grad_norm": 10.875,
"learning_rate": 1.271416039665657e-08,
"loss": 1.8871631622314453,
"step": 430
},
{
"epoch": 0.9248554913294798,
"grad_norm": 12.25,
"learning_rate": 7.905133346444464e-09,
"loss": 1.9046966552734375,
"step": 440
},
{
"epoch": 0.9458749343142406,
"grad_norm": 9.0,
"learning_rate": 4.219974526741527e-09,
"loss": 1.8698127746582032,
"step": 450
},
{
"epoch": 0.9668943772990016,
"grad_norm": 10.3125,
"learning_rate": 1.6757914168557264e-09,
"loss": 1.9108594894409179,
"step": 460
},
{
"epoch": 0.9879138202837625,
"grad_norm": 11.1875,
"learning_rate": 2.843947858848228e-10,
"loss": 1.9814664840698242,
"step": 470
},
{
"epoch": 1.0,
"step": 476,
"total_flos": 2.27705175303125e+17,
"train_loss": 1.8975331122133912,
"train_runtime": 3756.1832,
"train_samples_per_second": 0.507,
"train_steps_per_second": 0.127
}
],
"logging_steps": 10,
"max_steps": 476,
"num_input_tokens_seen": 0,
"num_train_epochs": 1,
"save_steps": 500,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": false,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 2.27705175303125e+17,
"train_batch_size": 1,
"trial_name": null,
"trial_params": null
}