Files
v37_v29_r0_perception_lr6e7…/trainer_state.json

352 lines
8.4 KiB
JSON
Raw Permalink Normal View History

{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 1.0,
"eval_steps": 500,
"global_step": 447,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.022396416573348264,
"grad_norm": 6.21875,
"learning_rate": 3.8571428571428574e-07,
"loss": 1.9765892028808594,
"step": 10
},
{
"epoch": 0.04479283314669653,
"grad_norm": 7.59375,
"learning_rate": 5.998026179790181e-07,
"loss": 1.9137596130371093,
"step": 20
},
{
"epoch": 0.0671892497200448,
"grad_norm": 15.8125,
"learning_rate": 5.982251198558757e-07,
"loss": 2.0828577041625977,
"step": 30
},
{
"epoch": 0.08958566629339305,
"grad_norm": 24.125,
"learning_rate": 5.950784240718929e-07,
"loss": 1.7728485107421874,
"step": 40
},
{
"epoch": 0.11198208286674133,
"grad_norm": 14.3125,
"learning_rate": 5.903790878763917e-07,
"loss": 1.9481042861938476,
"step": 50
},
{
"epoch": 0.1343784994400896,
"grad_norm": 11.9375,
"learning_rate": 5.841518381849623e-07,
"loss": 1.72081298828125,
"step": 60
},
{
"epoch": 0.15677491601343785,
"grad_norm": 6.3125,
"learning_rate": 5.764294414716493e-07,
"loss": 2.0641489028930664,
"step": 70
},
{
"epoch": 0.1791713325867861,
"grad_norm": 6.71875,
"learning_rate": 5.672525313586791e-07,
"loss": 1.887424850463867,
"step": 80
},
{
"epoch": 0.20156774916013437,
"grad_norm": 9.0625,
"learning_rate": 5.566693948109122e-07,
"loss": 1.74453125,
"step": 90
},
{
"epoch": 0.22396416573348266,
"grad_norm": 11.9375,
"learning_rate": 5.447357180600218e-07,
"loss": 2.011259078979492,
"step": 100
},
{
"epoch": 0.24636058230683092,
"grad_norm": 6.96875,
"learning_rate": 5.315142935952922e-07,
"loss": 1.8667875289916993,
"step": 110
},
{
"epoch": 0.2687569988801792,
"grad_norm": 4.90625,
"learning_rate": 5.170746897627867e-07,
"loss": 1.884227180480957,
"step": 120
},
{
"epoch": 0.2911534154535274,
"grad_norm": 8.625,
"learning_rate": 5.014928847113897e-07,
"loss": 1.7954404830932618,
"step": 130
},
{
"epoch": 0.3135498320268757,
"grad_norm": 20.875,
"learning_rate": 4.848508666118165e-07,
"loss": 1.8177595138549805,
"step": 140
},
{
"epoch": 0.335946248600224,
"grad_norm": 6.21875,
"learning_rate": 4.6723620225215703e-07,
"loss": 2.0977893829345704,
"step": 150
},
{
"epoch": 0.3583426651735722,
"grad_norm": 7.59375,
"learning_rate": 4.4874157627991176e-07,
"loss": 1.7599393844604492,
"step": 160
},
{
"epoch": 0.3807390817469205,
"grad_norm": 13.6875,
"learning_rate": 4.294643035149318e-07,
"loss": 2.047610855102539,
"step": 170
},
{
"epoch": 0.40313549832026874,
"grad_norm": 4.40625,
"learning_rate": 4.0950581689936744e-07,
"loss": 1.9894037246704102,
"step": 180
},
{
"epoch": 0.425531914893617,
"grad_norm": 6.09375,
"learning_rate": 3.8897113377892806e-07,
"loss": 1.8606115341186524,
"step": 190
},
{
"epoch": 0.4479283314669653,
"grad_norm": 4.0625,
"learning_rate": 3.679683033237648e-07,
"loss": 1.9338333129882812,
"step": 200
},
{
"epoch": 0.47032474804031354,
"grad_norm": 24.125,
"learning_rate": 3.4660783799653415e-07,
"loss": 2.0058767318725588,
"step": 210
},
{
"epoch": 0.49272116461366183,
"grad_norm": 14.75,
"learning_rate": 3.250021320591351e-07,
"loss": 2.109703254699707,
"step": 220
},
{
"epoch": 0.5151175811870101,
"grad_norm": 11.4375,
"learning_rate": 3.0326487017781626e-07,
"loss": 1.9790294647216797,
"step": 230
},
{
"epoch": 0.5375139977603584,
"grad_norm": 6.90625,
"learning_rate": 2.815104292384474e-07,
"loss": 1.7579475402832032,
"step": 240
},
{
"epoch": 0.5599104143337066,
"grad_norm": 6.0625,
"learning_rate": 2.5985327651947727e-07,
"loss": 2.0079017639160157,
"step": 250
},
{
"epoch": 0.5823068309070548,
"grad_norm": 8.3125,
"learning_rate": 2.3840736738926488e-07,
"loss": 1.9991819381713867,
"step": 260
},
{
"epoch": 0.6047032474804032,
"grad_norm": 4.40625,
"learning_rate": 2.172855456969734e-07,
"loss": 2.0415130615234376,
"step": 270
},
{
"epoch": 0.6270996640537514,
"grad_norm": 8.3125,
"learning_rate": 1.9659895001204434e-07,
"loss": 1.8622777938842774,
"step": 280
},
{
"epoch": 0.6494960806270996,
"grad_norm": 5.8125,
"learning_rate": 1.7645642883649244e-07,
"loss": 2.0226877212524412,
"step": 290
},
{
"epoch": 0.671892497200448,
"grad_norm": 4.21875,
"learning_rate": 1.5696396786705375e-07,
"loss": 1.9547592163085938,
"step": 300
},
{
"epoch": 0.6942889137737962,
"grad_norm": 13.1875,
"learning_rate": 1.3822413232081009e-07,
"loss": 1.98193302154541,
"step": 310
},
{
"epoch": 0.7166853303471444,
"grad_norm": 13.75,
"learning_rate": 1.2033552725865636e-07,
"loss": 1.8750802993774414,
"step": 320
},
{
"epoch": 0.7390817469204927,
"grad_norm": 13.0,
"learning_rate": 1.0339227874627297e-07,
"loss": 1.8019701004028321,
"step": 330
},
{
"epoch": 0.761478163493841,
"grad_norm": 17.375,
"learning_rate": 8.748353858262847e-08,
"loss": 1.6816850662231446,
"step": 340
},
{
"epoch": 0.7838745800671892,
"grad_norm": 13.9375,
"learning_rate": 7.269301520202152e-08,
"loss": 1.991793441772461,
"step": 350
},
{
"epoch": 0.8062709966405375,
"grad_norm": 4.5,
"learning_rate": 5.909853321796119e-08,
"loss": 1.8806726455688476,
"step": 360
},
{
"epoch": 0.8286674132138858,
"grad_norm": 9.5,
"learning_rate": 4.677162392646981e-08,
"loss": 2.013884925842285,
"step": 370
},
{
"epoch": 0.851063829787234,
"grad_norm": 8.6875,
"learning_rate": 3.577714892349537e-08,
"loss": 2.0343597412109373,
"step": 380
},
{
"epoch": 0.8734602463605823,
"grad_norm": 4.4375,
"learning_rate": 2.6172958816877822e-08,
"loss": 1.941878318786621,
"step": 390
},
{
"epoch": 0.8958566629339306,
"grad_norm": 8.6875,
"learning_rate": 1.800958882865644e-08,
"loss": 1.8717647552490235,
"step": 400
},
{
"epoch": 0.9182530795072789,
"grad_norm": 7.25,
"learning_rate": 1.1329992889392948e-08,
"loss": 1.965050506591797,
"step": 410
},
{
"epoch": 0.9406494960806271,
"grad_norm": 7.15625,
"learning_rate": 6.169317623651893e-09,
"loss": 1.9910579681396485,
"step": 420
},
{
"epoch": 0.9630459126539753,
"grad_norm": 5.375,
"learning_rate": 2.5547174158769502e-09,
"loss": 1.9617082595825195,
"step": 430
},
{
"epoch": 0.9854423292273237,
"grad_norm": 7.40625,
"learning_rate": 5.052115297494275e-10,
"loss": 1.8637035369873047,
"step": 440
},
{
"epoch": 1.0,
"step": 447,
"total_flos": 2.1268932852271104e+17,
"train_loss": 1.9281885245235708,
"train_runtime": 3538.1073,
"train_samples_per_second": 0.505,
"train_steps_per_second": 0.126
}
],
"logging_steps": 10,
"max_steps": 447,
"num_input_tokens_seen": 0,
"num_train_epochs": 1,
"save_steps": 500,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": false,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 2.1268932852271104e+17,
"train_batch_size": 1,
"trial_name": null,
"trial_params": null
}