1299 lines
29 KiB
JSON
1299 lines
29 KiB
JSON
[
|
|
{
|
|
"eval_loss": 5.064917087554932,
|
|
"eval_runtime": 82.8434,
|
|
"eval_samples_per_second": 6.337,
|
|
"eval_steps_per_second": 1.593,
|
|
"epoch": 0,
|
|
"step": 0
|
|
},
|
|
{
|
|
"loss": 4.4134,
|
|
"grad_norm": 34.756309509277344,
|
|
"learning_rate": 1.8e-05,
|
|
"epoch": 0.006222775357809583,
|
|
"step": 10
|
|
},
|
|
{
|
|
"loss": 2.8965,
|
|
"grad_norm": 16.411876678466797,
|
|
"learning_rate": 3.8e-05,
|
|
"epoch": 0.012445550715619166,
|
|
"step": 20
|
|
},
|
|
{
|
|
"loss": 2.1559,
|
|
"grad_norm": 10.711593627929688,
|
|
"learning_rate": 5.8e-05,
|
|
"epoch": 0.018668326073428748,
|
|
"step": 30
|
|
},
|
|
{
|
|
"loss": 2.0065,
|
|
"grad_norm": 9.655144691467285,
|
|
"learning_rate": 7.800000000000001e-05,
|
|
"epoch": 0.024891101431238332,
|
|
"step": 40
|
|
},
|
|
{
|
|
"loss": 1.9285,
|
|
"grad_norm": 10.14815902709961,
|
|
"learning_rate": 9.8e-05,
|
|
"epoch": 0.031113876789047916,
|
|
"step": 50
|
|
},
|
|
{
|
|
"loss": 1.905,
|
|
"grad_norm": 9.267602920532227,
|
|
"learning_rate": 0.000118,
|
|
"epoch": 0.037336652146857496,
|
|
"step": 60
|
|
},
|
|
{
|
|
"loss": 1.8057,
|
|
"grad_norm": 9.699400901794434,
|
|
"learning_rate": 0.000138,
|
|
"epoch": 0.043559427504667084,
|
|
"step": 70
|
|
},
|
|
{
|
|
"loss": 1.6726,
|
|
"grad_norm": 7.2970871925354,
|
|
"learning_rate": 0.00015800000000000002,
|
|
"epoch": 0.049782202862476664,
|
|
"step": 80
|
|
},
|
|
{
|
|
"eval_loss": 1.848530650138855,
|
|
"eval_runtime": 14.1069,
|
|
"eval_samples_per_second": 37.216,
|
|
"eval_steps_per_second": 9.357,
|
|
"epoch": 0.049782202862476664,
|
|
"step": 80
|
|
},
|
|
{
|
|
"loss": 1.7914,
|
|
"grad_norm": 7.156599998474121,
|
|
"learning_rate": 0.00017800000000000002,
|
|
"epoch": 0.056004978220286245,
|
|
"step": 90
|
|
},
|
|
{
|
|
"loss": 1.8229,
|
|
"grad_norm": 7.823815822601318,
|
|
"learning_rate": 0.00019800000000000002,
|
|
"epoch": 0.06222775357809583,
|
|
"step": 100
|
|
},
|
|
{
|
|
"loss": 1.9023,
|
|
"grad_norm": 8.658170700073242,
|
|
"learning_rate": 0.00019998239988424423,
|
|
"epoch": 0.06845052893590542,
|
|
"step": 110
|
|
},
|
|
{
|
|
"loss": 1.8214,
|
|
"grad_norm": 6.837601184844971,
|
|
"learning_rate": 0.0001999215679316913,
|
|
"epoch": 0.07467330429371499,
|
|
"step": 120
|
|
},
|
|
{
|
|
"loss": 1.8212,
|
|
"grad_norm": 6.444572448730469,
|
|
"learning_rate": 0.00019981731328627206,
|
|
"epoch": 0.08089607965152458,
|
|
"step": 130
|
|
},
|
|
{
|
|
"loss": 1.6969,
|
|
"grad_norm": 7.096024036407471,
|
|
"learning_rate": 0.00019966968125369522,
|
|
"epoch": 0.08711885500933417,
|
|
"step": 140
|
|
},
|
|
{
|
|
"loss": 1.7863,
|
|
"grad_norm": 6.71297550201416,
|
|
"learning_rate": 0.00019947873599008388,
|
|
"epoch": 0.09334163036714374,
|
|
"step": 150
|
|
},
|
|
{
|
|
"loss": 1.6954,
|
|
"grad_norm": 6.715601921081543,
|
|
"learning_rate": 0.00019924456047409517,
|
|
"epoch": 0.09956440572495333,
|
|
"step": 160
|
|
},
|
|
{
|
|
"eval_loss": 1.7422373294830322,
|
|
"eval_runtime": 13.8482,
|
|
"eval_samples_per_second": 37.911,
|
|
"eval_steps_per_second": 9.532,
|
|
"epoch": 0.09956440572495333,
|
|
"step": 160
|
|
},
|
|
{
|
|
"loss": 1.7892,
|
|
"grad_norm": 6.36358118057251,
|
|
"learning_rate": 0.00019896725647086072,
|
|
"epoch": 0.10578718108276292,
|
|
"step": 170
|
|
},
|
|
{
|
|
"loss": 1.7179,
|
|
"grad_norm": 8.09688949584961,
|
|
"learning_rate": 0.0001986469444877626,
|
|
"epoch": 0.11200995644057249,
|
|
"step": 180
|
|
},
|
|
{
|
|
"loss": 1.6861,
|
|
"grad_norm": 5.653729438781738,
|
|
"learning_rate": 0.0001982837637220647,
|
|
"epoch": 0.11823273179838208,
|
|
"step": 190
|
|
},
|
|
{
|
|
"loss": 1.6671,
|
|
"grad_norm": 6.865845203399658,
|
|
"learning_rate": 0.00019787787200042223,
|
|
"epoch": 0.12445550715619166,
|
|
"step": 200
|
|
},
|
|
{
|
|
"loss": 1.7291,
|
|
"grad_norm": 5.955836296081543,
|
|
"learning_rate": 0.00019742944571029517,
|
|
"epoch": 0.13067828251400124,
|
|
"step": 210
|
|
},
|
|
{
|
|
"loss": 1.626,
|
|
"grad_norm": 5.953787803649902,
|
|
"learning_rate": 0.00019693867972329598,
|
|
"epoch": 0.13690105787181084,
|
|
"step": 220
|
|
},
|
|
{
|
|
"loss": 1.5731,
|
|
"grad_norm": 5.453970909118652,
|
|
"learning_rate": 0.00019640578731050488,
|
|
"epoch": 0.1431238332296204,
|
|
"step": 230
|
|
},
|
|
{
|
|
"loss": 1.7516,
|
|
"grad_norm": 6.274188041687012,
|
|
"learning_rate": 0.00019583100004978886,
|
|
"epoch": 0.14934660858742999,
|
|
"step": 240
|
|
},
|
|
{
|
|
"eval_loss": 1.726481556892395,
|
|
"eval_runtime": 14.0063,
|
|
"eval_samples_per_second": 37.483,
|
|
"eval_steps_per_second": 9.424,
|
|
"epoch": 0.14934660858742999,
|
|
"step": 240
|
|
},
|
|
{
|
|
"loss": 1.603,
|
|
"grad_norm": 5.389549732208252,
|
|
"learning_rate": 0.00019521456772516552,
|
|
"epoch": 0.1555693839452396,
|
|
"step": 250
|
|
},
|
|
{
|
|
"loss": 1.6531,
|
|
"grad_norm": 5.605257987976074,
|
|
"learning_rate": 0.0001945567582182551,
|
|
"epoch": 0.16179215930304916,
|
|
"step": 260
|
|
},
|
|
{
|
|
"loss": 1.6709,
|
|
"grad_norm": 6.4445343017578125,
|
|
"learning_rate": 0.00019385785739186746,
|
|
"epoch": 0.16801493466085873,
|
|
"step": 270
|
|
},
|
|
{
|
|
"loss": 1.8325,
|
|
"grad_norm": 4.806204319000244,
|
|
"learning_rate": 0.0001931181689657756,
|
|
"epoch": 0.17423771001866833,
|
|
"step": 280
|
|
},
|
|
{
|
|
"loss": 1.7735,
|
|
"grad_norm": 5.748836994171143,
|
|
"learning_rate": 0.00019233801438472875,
|
|
"epoch": 0.1804604853764779,
|
|
"step": 290
|
|
},
|
|
{
|
|
"loss": 1.7027,
|
|
"grad_norm": 4.680629730224609,
|
|
"learning_rate": 0.00019151773267876273,
|
|
"epoch": 0.18668326073428748,
|
|
"step": 300
|
|
},
|
|
{
|
|
"loss": 1.5229,
|
|
"grad_norm": 5.180474758148193,
|
|
"learning_rate": 0.0001906576803158686,
|
|
"epoch": 0.19290603609209708,
|
|
"step": 310
|
|
},
|
|
{
|
|
"loss": 1.4272,
|
|
"grad_norm": 4.142552852630615,
|
|
"learning_rate": 0.00018975823104708313,
|
|
"epoch": 0.19912881144990666,
|
|
"step": 320
|
|
},
|
|
{
|
|
"eval_loss": 1.6637120246887207,
|
|
"eval_runtime": 14.0189,
|
|
"eval_samples_per_second": 37.45,
|
|
"eval_steps_per_second": 9.416,
|
|
"epoch": 0.19912881144990666,
|
|
"step": 320
|
|
},
|
|
{
|
|
"loss": 1.591,
|
|
"grad_norm": 5.900393486022949,
|
|
"learning_rate": 0.00018881977574406838,
|
|
"epoch": 0.20535158680771623,
|
|
"step": 330
|
|
},
|
|
{
|
|
"loss": 1.681,
|
|
"grad_norm": 5.877817153930664,
|
|
"learning_rate": 0.00018784272222925198,
|
|
"epoch": 0.21157436216552583,
|
|
"step": 340
|
|
},
|
|
{
|
|
"loss": 1.5724,
|
|
"grad_norm": 4.48579216003418,
|
|
"learning_rate": 0.00018682749509860012,
|
|
"epoch": 0.2177971375233354,
|
|
"step": 350
|
|
},
|
|
{
|
|
"loss": 1.6694,
|
|
"grad_norm": 6.317138195037842,
|
|
"learning_rate": 0.00018577453553710215,
|
|
"epoch": 0.22401991288114498,
|
|
"step": 360
|
|
},
|
|
{
|
|
"loss": 1.585,
|
|
"grad_norm": 5.043068885803223,
|
|
"learning_rate": 0.00018468430112704573,
|
|
"epoch": 0.23024268823895458,
|
|
"step": 370
|
|
},
|
|
{
|
|
"loss": 1.4904,
|
|
"grad_norm": 4.478468894958496,
|
|
"learning_rate": 0.00018355726564916628,
|
|
"epoch": 0.23646546359676415,
|
|
"step": 380
|
|
},
|
|
{
|
|
"loss": 1.5321,
|
|
"grad_norm": 6.474656105041504,
|
|
"learning_rate": 0.00018239391887675722,
|
|
"epoch": 0.24268823895457373,
|
|
"step": 390
|
|
},
|
|
{
|
|
"loss": 1.5504,
|
|
"grad_norm": 5.6627726554870605,
|
|
"learning_rate": 0.00018119476636283018,
|
|
"epoch": 0.24891101431238333,
|
|
"step": 400
|
|
},
|
|
{
|
|
"eval_loss": 1.6094003915786743,
|
|
"eval_runtime": 13.9871,
|
|
"eval_samples_per_second": 37.535,
|
|
"eval_steps_per_second": 9.437,
|
|
"epoch": 0.24891101431238333,
|
|
"step": 400
|
|
},
|
|
{
|
|
"loss": 1.5178,
|
|
"grad_norm": 3.671290397644043,
|
|
"learning_rate": 0.00017996032922041797,
|
|
"epoch": 0.2551337896701929,
|
|
"step": 410
|
|
},
|
|
{
|
|
"loss": 1.5326,
|
|
"grad_norm": 4.7752814292907715,
|
|
"learning_rate": 0.00017869114389611575,
|
|
"epoch": 0.2613565650280025,
|
|
"step": 420
|
|
},
|
|
{
|
|
"loss": 1.6097,
|
|
"grad_norm": 4.629915237426758,
|
|
"learning_rate": 0.00017738776193695853,
|
|
"epoch": 0.26757934038581205,
|
|
"step": 430
|
|
},
|
|
{
|
|
"loss": 1.5883,
|
|
"grad_norm": 4.744765281677246,
|
|
"learning_rate": 0.00017605074975073664,
|
|
"epoch": 0.2738021157436217,
|
|
"step": 440
|
|
},
|
|
{
|
|
"loss": 1.5456,
|
|
"grad_norm": 5.553370952606201,
|
|
"learning_rate": 0.00017468068835985325,
|
|
"epoch": 0.28002489110143125,
|
|
"step": 450
|
|
},
|
|
{
|
|
"loss": 1.635,
|
|
"grad_norm": 4.544879913330078,
|
|
"learning_rate": 0.00017327817314883055,
|
|
"epoch": 0.2862476664592408,
|
|
"step": 460
|
|
},
|
|
{
|
|
"loss": 1.3816,
|
|
"grad_norm": 4.572506427764893,
|
|
"learning_rate": 0.00017184381360557498,
|
|
"epoch": 0.2924704418170504,
|
|
"step": 470
|
|
},
|
|
{
|
|
"loss": 1.559,
|
|
"grad_norm": 5.5629119873046875,
|
|
"learning_rate": 0.00017037823305651343,
|
|
"epoch": 0.29869321717485997,
|
|
"step": 480
|
|
},
|
|
{
|
|
"eval_loss": 1.573219656944275,
|
|
"eval_runtime": 13.9728,
|
|
"eval_samples_per_second": 37.573,
|
|
"eval_steps_per_second": 9.447,
|
|
"epoch": 0.29869321717485997,
|
|
"step": 480
|
|
},
|
|
{
|
|
"loss": 1.5028,
|
|
"grad_norm": 4.841222763061523,
|
|
"learning_rate": 0.0001688820683957156,
|
|
"epoch": 0.30491599253266954,
|
|
"step": 490
|
|
},
|
|
{
|
|
"loss": 1.5322,
|
|
"grad_norm": 4.653014183044434,
|
|
"learning_rate": 0.00016735596980812047,
|
|
"epoch": 0.3111387678904792,
|
|
"step": 500
|
|
},
|
|
{
|
|
"loss": 1.5097,
|
|
"grad_norm": 5.121198654174805,
|
|
"learning_rate": 0.0001658006004869867,
|
|
"epoch": 0.31736154324828875,
|
|
"step": 510
|
|
},
|
|
{
|
|
"loss": 1.6415,
|
|
"grad_norm": 5.429177761077881,
|
|
"learning_rate": 0.00016421663634569046,
|
|
"epoch": 0.3235843186060983,
|
|
"step": 520
|
|
},
|
|
{
|
|
"loss": 1.5053,
|
|
"grad_norm": 6.6233673095703125,
|
|
"learning_rate": 0.00016260476572399496,
|
|
"epoch": 0.3298070939639079,
|
|
"step": 530
|
|
},
|
|
{
|
|
"loss": 1.4254,
|
|
"grad_norm": 4.580263137817383,
|
|
"learning_rate": 0.00016096568908892047,
|
|
"epoch": 0.33602986932171747,
|
|
"step": 540
|
|
},
|
|
{
|
|
"loss": 1.5475,
|
|
"grad_norm": 5.61928129196167,
|
|
"learning_rate": 0.00015930011873034375,
|
|
"epoch": 0.3422526446795271,
|
|
"step": 550
|
|
},
|
|
{
|
|
"loss": 1.4719,
|
|
"grad_norm": 5.081675052642822,
|
|
"learning_rate": 0.00015760877845145995,
|
|
"epoch": 0.34847542003733667,
|
|
"step": 560
|
|
},
|
|
{
|
|
"eval_loss": 1.572365403175354,
|
|
"eval_runtime": 14.0923,
|
|
"eval_samples_per_second": 37.254,
|
|
"eval_steps_per_second": 9.367,
|
|
"epoch": 0.34847542003733667,
|
|
"step": 560
|
|
},
|
|
{
|
|
"loss": 1.5566,
|
|
"grad_norm": 4.780381679534912,
|
|
"learning_rate": 0.00015589240325424088,
|
|
"epoch": 0.35469819539514624,
|
|
"step": 570
|
|
},
|
|
{
|
|
"loss": 1.5206,
|
|
"grad_norm": 4.883270740509033,
|
|
"learning_rate": 0.00015415173902002703,
|
|
"epoch": 0.3609209707529558,
|
|
"step": 580
|
|
},
|
|
{
|
|
"loss": 1.3941,
|
|
"grad_norm": 6.177889823913574,
|
|
"learning_rate": 0.00015238754218539156,
|
|
"epoch": 0.3671437461107654,
|
|
"step": 590
|
|
},
|
|
{
|
|
"loss": 1.4619,
|
|
"grad_norm": 4.226222038269043,
|
|
"learning_rate": 0.00015060057941341718,
|
|
"epoch": 0.37336652146857496,
|
|
"step": 600
|
|
},
|
|
{
|
|
"loss": 1.3478,
|
|
"grad_norm": 5.7228102684021,
|
|
"learning_rate": 0.00014879162726052928,
|
|
"epoch": 0.3795892968263846,
|
|
"step": 610
|
|
},
|
|
{
|
|
"loss": 1.5298,
|
|
"grad_norm": 4.754924774169922,
|
|
"learning_rate": 0.0001469614718390295,
|
|
"epoch": 0.38581207218419417,
|
|
"step": 620
|
|
},
|
|
{
|
|
"loss": 1.4809,
|
|
"grad_norm": 5.708168983459473,
|
|
"learning_rate": 0.00014511090847547643,
|
|
"epoch": 0.39203484754200374,
|
|
"step": 630
|
|
},
|
|
{
|
|
"loss": 1.5381,
|
|
"grad_norm": 5.013388633728027,
|
|
"learning_rate": 0.00014324074136506284,
|
|
"epoch": 0.3982576228998133,
|
|
"step": 640
|
|
},
|
|
{
|
|
"eval_loss": 1.5396426916122437,
|
|
"eval_runtime": 13.9405,
|
|
"eval_samples_per_second": 37.66,
|
|
"eval_steps_per_second": 9.469,
|
|
"epoch": 0.3982576228998133,
|
|
"step": 640
|
|
},
|
|
{
|
|
"loss": 1.4894,
|
|
"grad_norm": 5.063615322113037,
|
|
"learning_rate": 0.00014135178322213765,
|
|
"epoch": 0.4044803982576229,
|
|
"step": 650
|
|
},
|
|
{
|
|
"loss": 1.5547,
|
|
"grad_norm": 4.9530744552612305,
|
|
"learning_rate": 0.00013944485492702716,
|
|
"epoch": 0.41070317361543246,
|
|
"step": 660
|
|
},
|
|
{
|
|
"loss": 1.4292,
|
|
"grad_norm": 4.819840908050537,
|
|
"learning_rate": 0.00013752078516930652,
|
|
"epoch": 0.4169259489732421,
|
|
"step": 670
|
|
},
|
|
{
|
|
"loss": 1.556,
|
|
"grad_norm": 4.666694641113281,
|
|
"learning_rate": 0.00013558041008767798,
|
|
"epoch": 0.42314872433105166,
|
|
"step": 680
|
|
},
|
|
{
|
|
"loss": 1.5216,
|
|
"grad_norm": 5.643951416015625,
|
|
"learning_rate": 0.00013362457290661215,
|
|
"epoch": 0.42937149968886124,
|
|
"step": 690
|
|
},
|
|
{
|
|
"loss": 1.5768,
|
|
"grad_norm": 6.549987316131592,
|
|
"learning_rate": 0.00013165412356990955,
|
|
"epoch": 0.4355942750466708,
|
|
"step": 700
|
|
},
|
|
{
|
|
"loss": 1.4482,
|
|
"grad_norm": 5.201975345611572,
|
|
"learning_rate": 0.0001296699183713427,
|
|
"epoch": 0.4418170504044804,
|
|
"step": 710
|
|
},
|
|
{
|
|
"loss": 1.4855,
|
|
"grad_norm": 4.325445652008057,
|
|
"learning_rate": 0.0001276728195825383,
|
|
"epoch": 0.44803982576228996,
|
|
"step": 720
|
|
},
|
|
{
|
|
"eval_loss": 1.5151004791259766,
|
|
"eval_runtime": 13.9174,
|
|
"eval_samples_per_second": 37.723,
|
|
"eval_steps_per_second": 9.485,
|
|
"epoch": 0.44803982576228996,
|
|
"step": 720
|
|
},
|
|
{
|
|
"loss": 1.4242,
|
|
"grad_norm": 6.4271931648254395,
|
|
"learning_rate": 0.00012566369507826175,
|
|
"epoch": 0.4542626011200996,
|
|
"step": 730
|
|
},
|
|
{
|
|
"loss": 1.544,
|
|
"grad_norm": 4.974382400512695,
|
|
"learning_rate": 0.00012364341795926683,
|
|
"epoch": 0.46048537647790916,
|
|
"step": 740
|
|
},
|
|
{
|
|
"loss": 1.3544,
|
|
"grad_norm": 4.475541114807129,
|
|
"learning_rate": 0.00012161286617287419,
|
|
"epoch": 0.46670815183571873,
|
|
"step": 750
|
|
},
|
|
{
|
|
"loss": 1.5565,
|
|
"grad_norm": 4.763009548187256,
|
|
"learning_rate": 0.00011957292213144385,
|
|
"epoch": 0.4729309271935283,
|
|
"step": 760
|
|
},
|
|
{
|
|
"loss": 1.4961,
|
|
"grad_norm": 5.890961647033691,
|
|
"learning_rate": 0.00011752447232890702,
|
|
"epoch": 0.4791537025513379,
|
|
"step": 770
|
|
},
|
|
{
|
|
"loss": 1.4762,
|
|
"grad_norm": 5.207292556762695,
|
|
"learning_rate": 0.00011546840695552466,
|
|
"epoch": 0.48537647790914745,
|
|
"step": 780
|
|
},
|
|
{
|
|
"loss": 1.5207,
|
|
"grad_norm": 4.962773323059082,
|
|
"learning_rate": 0.0001134056195110393,
|
|
"epoch": 0.4915992532669571,
|
|
"step": 790
|
|
},
|
|
{
|
|
"loss": 1.3912,
|
|
"grad_norm": 4.377303123474121,
|
|
"learning_rate": 0.00011133700641638891,
|
|
"epoch": 0.49782202862476665,
|
|
"step": 800
|
|
},
|
|
{
|
|
"eval_loss": 1.4861326217651367,
|
|
"eval_runtime": 13.9515,
|
|
"eval_samples_per_second": 37.63,
|
|
"eval_steps_per_second": 9.461,
|
|
"epoch": 0.49782202862476665,
|
|
"step": 800
|
|
},
|
|
{
|
|
"loss": 1.3795,
|
|
"grad_norm": 4.376067161560059,
|
|
"learning_rate": 0.0001092634666241513,
|
|
"epoch": 0.5040448039825762,
|
|
"step": 810
|
|
},
|
|
{
|
|
"loss": 1.4531,
|
|
"grad_norm": 4.802002429962158,
|
|
"learning_rate": 0.00010718590122788821,
|
|
"epoch": 0.5102675793403858,
|
|
"step": 820
|
|
},
|
|
{
|
|
"loss": 1.435,
|
|
"grad_norm": 4.865792751312256,
|
|
"learning_rate": 0.00010510521307055914,
|
|
"epoch": 0.5164903546981954,
|
|
"step": 830
|
|
},
|
|
{
|
|
"loss": 1.4412,
|
|
"grad_norm": 5.106067180633545,
|
|
"learning_rate": 0.000103022306352175,
|
|
"epoch": 0.522713130056005,
|
|
"step": 840
|
|
},
|
|
{
|
|
"loss": 1.398,
|
|
"grad_norm": 4.740355491638184,
|
|
"learning_rate": 0.00010093808623686165,
|
|
"epoch": 0.5289359054138145,
|
|
"step": 850
|
|
},
|
|
{
|
|
"loss": 1.5138,
|
|
"grad_norm": 4.870340347290039,
|
|
"learning_rate": 9.88534584595051e-05,
|
|
"epoch": 0.5351586807716241,
|
|
"step": 860
|
|
},
|
|
{
|
|
"loss": 1.4865,
|
|
"grad_norm": 4.641928195953369,
|
|
"learning_rate": 9.676932893214805e-05,
|
|
"epoch": 0.5413814561294338,
|
|
"step": 870
|
|
},
|
|
{
|
|
"loss": 1.3906,
|
|
"grad_norm": 6.179795742034912,
|
|
"learning_rate": 9.46866033503098e-05,
|
|
"epoch": 0.5476042314872434,
|
|
"step": 880
|
|
},
|
|
{
|
|
"eval_loss": 1.473583459854126,
|
|
"eval_runtime": 13.9209,
|
|
"eval_samples_per_second": 37.713,
|
|
"eval_steps_per_second": 9.482,
|
|
"epoch": 0.5476042314872434,
|
|
"step": 880
|
|
},
|
|
{
|
|
"loss": 1.5219,
|
|
"grad_norm": 5.138612270355225,
|
|
"learning_rate": 9.260618679940025e-05,
|
|
"epoch": 0.5538270068450529,
|
|
"step": 890
|
|
},
|
|
{
|
|
"loss": 1.4397,
|
|
"grad_norm": 4.696632385253906,
|
|
"learning_rate": 9.05289833613988e-05,
|
|
"epoch": 0.5600497822028625,
|
|
"step": 900
|
|
},
|
|
{
|
|
"loss": 1.5194,
|
|
"grad_norm": 5.26527738571167,
|
|
"learning_rate": 8.845589572196961e-05,
|
|
"epoch": 0.5662725575606721,
|
|
"step": 910
|
|
},
|
|
{
|
|
"loss": 1.388,
|
|
"grad_norm": 4.302639961242676,
|
|
"learning_rate": 8.638782477818334e-05,
|
|
"epoch": 0.5724953329184816,
|
|
"step": 920
|
|
},
|
|
{
|
|
"loss": 1.3796,
|
|
"grad_norm": 5.462881565093994,
|
|
"learning_rate": 8.432566924701659e-05,
|
|
"epoch": 0.5787181082762912,
|
|
"step": 930
|
|
},
|
|
{
|
|
"loss": 1.4552,
|
|
"grad_norm": 5.315045356750488,
|
|
"learning_rate": 8.227032527479806e-05,
|
|
"epoch": 0.5849408836341008,
|
|
"step": 940
|
|
},
|
|
{
|
|
"loss": 1.3473,
|
|
"grad_norm": 4.484452247619629,
|
|
"learning_rate": 8.022268604777271e-05,
|
|
"epoch": 0.5911636589919104,
|
|
"step": 950
|
|
},
|
|
{
|
|
"loss": 1.4053,
|
|
"grad_norm": 4.7990312576293945,
|
|
"learning_rate": 7.818364140395137e-05,
|
|
"epoch": 0.5973864343497199,
|
|
"step": 960
|
|
},
|
|
{
|
|
"eval_loss": 1.4610518217086792,
|
|
"eval_runtime": 14.0919,
|
|
"eval_samples_per_second": 37.255,
|
|
"eval_steps_per_second": 9.367,
|
|
"epoch": 0.5973864343497199,
|
|
"step": 960
|
|
},
|
|
{
|
|
"loss": 1.4184,
|
|
"grad_norm": 5.96822452545166,
|
|
"learning_rate": 7.615407744641619e-05,
|
|
"epoch": 0.6036092097075295,
|
|
"step": 970
|
|
},
|
|
{
|
|
"loss": 1.495,
|
|
"grad_norm": 4.883596420288086,
|
|
"learning_rate": 7.413487615824847e-05,
|
|
"epoch": 0.6098319850653391,
|
|
"step": 980
|
|
},
|
|
{
|
|
"loss": 1.4563,
|
|
"grad_norm": 4.199883460998535,
|
|
"learning_rate": 7.212691501924753e-05,
|
|
"epoch": 0.6160547604231488,
|
|
"step": 990
|
|
},
|
|
{
|
|
"loss": 1.4791,
|
|
"grad_norm": 4.85598611831665,
|
|
"learning_rate": 7.013106662460604e-05,
|
|
"epoch": 0.6222775357809583,
|
|
"step": 1000
|
|
},
|
|
{
|
|
"loss": 1.4621,
|
|
"grad_norm": 5.302894115447998,
|
|
"learning_rate": 6.81481983057085e-05,
|
|
"epoch": 0.6285003111387679,
|
|
"step": 1010
|
|
},
|
|
{
|
|
"loss": 1.3599,
|
|
"grad_norm": 4.943319797515869,
|
|
"learning_rate": 6.617917175321669e-05,
|
|
"epoch": 0.6347230864965775,
|
|
"step": 1020
|
|
},
|
|
{
|
|
"loss": 1.2883,
|
|
"grad_norm": 6.421422958374023,
|
|
"learning_rate": 6.422484264260698e-05,
|
|
"epoch": 0.6409458618543871,
|
|
"step": 1030
|
|
},
|
|
{
|
|
"loss": 1.392,
|
|
"grad_norm": 4.642984867095947,
|
|
"learning_rate": 6.228606026232118e-05,
|
|
"epoch": 0.6471686372121966,
|
|
"step": 1040
|
|
},
|
|
{
|
|
"eval_loss": 1.450961709022522,
|
|
"eval_runtime": 13.9604,
|
|
"eval_samples_per_second": 37.606,
|
|
"eval_steps_per_second": 9.455,
|
|
"epoch": 0.6471686372121966,
|
|
"step": 1040
|
|
},
|
|
{
|
|
"loss": 1.3627,
|
|
"grad_norm": 4.882229328155518,
|
|
"learning_rate": 6.0363667144693105e-05,
|
|
"epoch": 0.6533914125700062,
|
|
"step": 1050
|
|
},
|
|
{
|
|
"loss": 1.4695,
|
|
"grad_norm": 4.268496990203857,
|
|
"learning_rate": 5.845849869981137e-05,
|
|
"epoch": 0.6596141879278158,
|
|
"step": 1060
|
|
},
|
|
{
|
|
"loss": 1.2919,
|
|
"grad_norm": 4.607876777648926,
|
|
"learning_rate": 5.657138285247687e-05,
|
|
"epoch": 0.6658369632856254,
|
|
"step": 1070
|
|
},
|
|
{
|
|
"loss": 1.3095,
|
|
"grad_norm": 5.3004937171936035,
|
|
"learning_rate": 5.4703139682413586e-05,
|
|
"epoch": 0.6720597386434349,
|
|
"step": 1080
|
|
},
|
|
{
|
|
"loss": 1.4059,
|
|
"grad_norm": 4.090771675109863,
|
|
"learning_rate": 5.285458106788807e-05,
|
|
"epoch": 0.6782825140012445,
|
|
"step": 1090
|
|
},
|
|
{
|
|
"loss": 1.2986,
|
|
"grad_norm": 4.546812057495117,
|
|
"learning_rate": 5.10265103328937e-05,
|
|
"epoch": 0.6845052893590542,
|
|
"step": 1100
|
|
},
|
|
{
|
|
"loss": 1.3874,
|
|
"grad_norm": 4.813508987426758,
|
|
"learning_rate": 4.921972189805154e-05,
|
|
"epoch": 0.6907280647168638,
|
|
"step": 1110
|
|
},
|
|
{
|
|
"loss": 1.5213,
|
|
"grad_norm": 4.514693737030029,
|
|
"learning_rate": 4.7435000935381115e-05,
|
|
"epoch": 0.6969508400746733,
|
|
"step": 1120
|
|
},
|
|
{
|
|
"eval_loss": 1.4407389163970947,
|
|
"eval_runtime": 14.4277,
|
|
"eval_samples_per_second": 36.388,
|
|
"eval_steps_per_second": 9.149,
|
|
"epoch": 0.6969508400746733,
|
|
"step": 1120
|
|
},
|
|
{
|
|
"loss": 1.3087,
|
|
"grad_norm": 4.339325428009033,
|
|
"learning_rate": 4.567312302708965e-05,
|
|
"epoch": 0.7031736154324829,
|
|
"step": 1130
|
|
},
|
|
{
|
|
"loss": 1.5258,
|
|
"grad_norm": 7.012087821960449,
|
|
"learning_rate": 4.393485382852935e-05,
|
|
"epoch": 0.7093963907902925,
|
|
"step": 1140
|
|
},
|
|
{
|
|
"loss": 1.2682,
|
|
"grad_norm": 4.673589706420898,
|
|
"learning_rate": 4.2220948735467967e-05,
|
|
"epoch": 0.7156191661481021,
|
|
"step": 1150
|
|
},
|
|
{
|
|
"loss": 1.3627,
|
|
"grad_norm": 5.091920852661133,
|
|
"learning_rate": 4.053215255581844e-05,
|
|
"epoch": 0.7218419415059116,
|
|
"step": 1160
|
|
},
|
|
{
|
|
"loss": 1.5111,
|
|
"grad_norm": 4.602088928222656,
|
|
"learning_rate": 3.886919918596894e-05,
|
|
"epoch": 0.7280647168637212,
|
|
"step": 1170
|
|
},
|
|
{
|
|
"loss": 1.3743,
|
|
"grad_norm": 4.942876815795898,
|
|
"learning_rate": 3.723281129185574e-05,
|
|
"epoch": 0.7342874922215308,
|
|
"step": 1180
|
|
},
|
|
{
|
|
"loss": 1.3959,
|
|
"grad_norm": 4.139978885650635,
|
|
"learning_rate": 3.562369999491536e-05,
|
|
"epoch": 0.7405102675793404,
|
|
"step": 1190
|
|
},
|
|
{
|
|
"loss": 1.4383,
|
|
"grad_norm": 5.013376235961914,
|
|
"learning_rate": 3.4042564563054526e-05,
|
|
"epoch": 0.7467330429371499,
|
|
"step": 1200
|
|
},
|
|
{
|
|
"eval_loss": 1.4313360452651978,
|
|
"eval_runtime": 14.3246,
|
|
"eval_samples_per_second": 36.65,
|
|
"eval_steps_per_second": 9.215,
|
|
"epoch": 0.7467330429371499,
|
|
"step": 1200
|
|
},
|
|
{
|
|
"loss": 1.3621,
|
|
"grad_norm": 5.042903900146484,
|
|
"learning_rate": 3.249009210677054e-05,
|
|
"epoch": 0.7529558182949595,
|
|
"step": 1210
|
|
},
|
|
{
|
|
"loss": 1.4528,
|
|
"grad_norm": 4.191718101501465,
|
|
"learning_rate": 3.096695728055536e-05,
|
|
"epoch": 0.7591785936527692,
|
|
"step": 1220
|
|
},
|
|
{
|
|
"loss": 1.326,
|
|
"grad_norm": 4.906574249267578,
|
|
"learning_rate": 2.9473821989712625e-05,
|
|
"epoch": 0.7654013690105788,
|
|
"step": 1230
|
|
},
|
|
{
|
|
"loss": 1.3792,
|
|
"grad_norm": 5.21110200881958,
|
|
"learning_rate": 2.801133510271463e-05,
|
|
"epoch": 0.7716241443683883,
|
|
"step": 1240
|
|
},
|
|
{
|
|
"loss": 1.3634,
|
|
"grad_norm": 5.0998430252075195,
|
|
"learning_rate": 2.6580132169225335e-05,
|
|
"epoch": 0.7778469197261979,
|
|
"step": 1250
|
|
},
|
|
{
|
|
"loss": 1.378,
|
|
"grad_norm": 5.857067108154297,
|
|
"learning_rate": 2.5180835143910732e-05,
|
|
"epoch": 0.7840696950840075,
|
|
"step": 1260
|
|
},
|
|
{
|
|
"loss": 1.2895,
|
|
"grad_norm": 4.248624801635742,
|
|
"learning_rate": 2.3814052116157492e-05,
|
|
"epoch": 0.790292470441817,
|
|
"step": 1270
|
|
},
|
|
{
|
|
"loss": 1.3369,
|
|
"grad_norm": 4.634941577911377,
|
|
"learning_rate": 2.248037704581686e-05,
|
|
"epoch": 0.7965152457996266,
|
|
"step": 1280
|
|
},
|
|
{
|
|
"eval_loss": 1.4184999465942383,
|
|
"eval_runtime": 14.0409,
|
|
"eval_samples_per_second": 37.391,
|
|
"eval_steps_per_second": 9.401,
|
|
"epoch": 0.7965152457996266,
|
|
"step": 1280
|
|
},
|
|
{
|
|
"loss": 1.2569,
|
|
"grad_norm": 4.819410800933838,
|
|
"learning_rate": 2.1180389505089004e-05,
|
|
"epoch": 0.8027380211574362,
|
|
"step": 1290
|
|
},
|
|
{
|
|
"loss": 1.2557,
|
|
"grad_norm": 4.907857894897461,
|
|
"learning_rate": 1.9914654426659374e-05,
|
|
"epoch": 0.8089607965152458,
|
|
"step": 1300
|
|
},
|
|
{
|
|
"loss": 1.4703,
|
|
"grad_norm": 4.4456000328063965,
|
|
"learning_rate": 1.8683721858197366e-05,
|
|
"epoch": 0.8151835718730553,
|
|
"step": 1310
|
|
},
|
|
{
|
|
"loss": 1.359,
|
|
"grad_norm": 5.10971736907959,
|
|
"learning_rate": 1.7488126723323183e-05,
|
|
"epoch": 0.8214063472308649,
|
|
"step": 1320
|
|
},
|
|
{
|
|
"loss": 1.3467,
|
|
"grad_norm": 4.865292549133301,
|
|
"learning_rate": 1.632838858914747e-05,
|
|
"epoch": 0.8276291225886746,
|
|
"step": 1330
|
|
},
|
|
{
|
|
"loss": 1.41,
|
|
"grad_norm": 4.803621292114258,
|
|
"learning_rate": 1.5205011440483929e-05,
|
|
"epoch": 0.8338518979464842,
|
|
"step": 1340
|
|
},
|
|
{
|
|
"loss": 1.4062,
|
|
"grad_norm": 4.7581377029418945,
|
|
"learning_rate": 1.4118483460834064e-05,
|
|
"epoch": 0.8400746733042938,
|
|
"step": 1350
|
|
},
|
|
{
|
|
"loss": 1.375,
|
|
"grad_norm": 5.33804988861084,
|
|
"learning_rate": 1.3069276820237997e-05,
|
|
"epoch": 0.8462974486621033,
|
|
"step": 1360
|
|
},
|
|
{
|
|
"eval_loss": 1.4135597944259644,
|
|
"eval_runtime": 14.1252,
|
|
"eval_samples_per_second": 37.168,
|
|
"eval_steps_per_second": 9.345,
|
|
"epoch": 0.8462974486621033,
|
|
"step": 1360
|
|
},
|
|
{
|
|
"loss": 1.4474,
|
|
"grad_norm": 4.493839263916016,
|
|
"learning_rate": 1.2057847470084993e-05,
|
|
"epoch": 0.8525202240199129,
|
|
"step": 1370
|
|
},
|
|
{
|
|
"loss": 1.3765,
|
|
"grad_norm": 3.7897987365722656,
|
|
"learning_rate": 1.108463494497135e-05,
|
|
"epoch": 0.8587429993777225,
|
|
"step": 1380
|
|
},
|
|
{
|
|
"loss": 1.3583,
|
|
"grad_norm": 3.9571006298065186,
|
|
"learning_rate": 1.0150062171693076e-05,
|
|
"epoch": 0.864965774735532,
|
|
"step": 1390
|
|
},
|
|
{
|
|
"loss": 1.4705,
|
|
"grad_norm": 4.610692501068115,
|
|
"learning_rate": 9.254535285455334e-06,
|
|
"epoch": 0.8711885500933416,
|
|
"step": 1400
|
|
},
|
|
{
|
|
"loss": 1.3517,
|
|
"grad_norm": 4.550515651702881,
|
|
"learning_rate": 8.398443453379267e-06,
|
|
"epoch": 0.8774113254511512,
|
|
"step": 1410
|
|
},
|
|
{
|
|
"loss": 1.2454,
|
|
"grad_norm": 4.898622512817383,
|
|
"learning_rate": 7.582158705382581e-06,
|
|
"epoch": 0.8836341008089608,
|
|
"step": 1420
|
|
},
|
|
{
|
|
"loss": 1.3031,
|
|
"grad_norm": 4.423801422119141,
|
|
"learning_rate": 6.806035772507169e-06,
|
|
"epoch": 0.8898568761667703,
|
|
"step": 1430
|
|
},
|
|
{
|
|
"loss": 1.3421,
|
|
"grad_norm": 3.7221221923828125,
|
|
"learning_rate": 6.070411932764586e-06,
|
|
"epoch": 0.8960796515245799,
|
|
"step": 1440
|
|
},
|
|
{
|
|
"eval_loss": 1.4087599515914917,
|
|
"eval_runtime": 14.0409,
|
|
"eval_samples_per_second": 37.391,
|
|
"eval_steps_per_second": 9.401,
|
|
"epoch": 0.8960796515245799,
|
|
"step": 1440
|
|
},
|
|
{
|
|
"loss": 1.3722,
|
|
"grad_norm": 5.566176414489746,
|
|
"learning_rate": 5.375606864565785e-06,
|
|
"epoch": 0.9023024268823896,
|
|
"step": 1450
|
|
},
|
|
{
|
|
"loss": 1.4635,
|
|
"grad_norm": 5.002288818359375,
|
|
"learning_rate": 4.721922507799248e-06,
|
|
"epoch": 0.9085252022401992,
|
|
"step": 1460
|
|
},
|
|
{
|
|
"loss": 1.2757,
|
|
"grad_norm": 3.9484496116638184,
|
|
"learning_rate": 4.10964293261763e-06,
|
|
"epoch": 0.9147479775980087,
|
|
"step": 1470
|
|
},
|
|
{
|
|
"loss": 1.3995,
|
|
"grad_norm": 5.582644939422607,
|
|
"learning_rate": 3.5390342159900223e-06,
|
|
"epoch": 0.9209707529558183,
|
|
"step": 1480
|
|
},
|
|
{
|
|
"loss": 1.3435,
|
|
"grad_norm": 4.107953071594238,
|
|
"learning_rate": 3.0103443260734554e-06,
|
|
"epoch": 0.9271935283136279,
|
|
"step": 1490
|
|
},
|
|
{
|
|
"loss": 1.4009,
|
|
"grad_norm": 4.991889953613281,
|
|
"learning_rate": 2.5238030144539737e-06,
|
|
"epoch": 0.9334163036714375,
|
|
"step": 1500
|
|
},
|
|
{
|
|
"loss": 1.4222,
|
|
"grad_norm": 4.622715473175049,
|
|
"learning_rate": 2.079621716303959e-06,
|
|
"epoch": 0.939639079029247,
|
|
"step": 1510
|
|
},
|
|
{
|
|
"loss": 1.3094,
|
|
"grad_norm": 3.724646806716919,
|
|
"learning_rate": 1.6779934584992718e-06,
|
|
"epoch": 0.9458618543870566,
|
|
"step": 1520
|
|
},
|
|
{
|
|
"eval_loss": 1.408400058746338,
|
|
"eval_runtime": 14.2644,
|
|
"eval_samples_per_second": 36.805,
|
|
"eval_steps_per_second": 9.254,
|
|
"epoch": 0.9458618543870566,
|
|
"step": 1520
|
|
},
|
|
{
|
|
"loss": 1.2012,
|
|
"grad_norm": 5.017258644104004,
|
|
"learning_rate": 1.3190927757358973e-06,
|
|
"epoch": 0.9520846297448662,
|
|
"step": 1530
|
|
},
|
|
{
|
|
"loss": 1.4522,
|
|
"grad_norm": 5.213080883026123,
|
|
"learning_rate": 1.0030756346829151e-06,
|
|
"epoch": 0.9583074051026758,
|
|
"step": 1540
|
|
},
|
|
{
|
|
"loss": 1.4536,
|
|
"grad_norm": 4.179771900177002,
|
|
"learning_rate": 7.300793662043282e-07,
|
|
"epoch": 0.9645301804604853,
|
|
"step": 1550
|
|
},
|
|
{
|
|
"loss": 1.3803,
|
|
"grad_norm": 4.139328479766846,
|
|
"learning_rate": 5.002226056795123e-07,
|
|
"epoch": 0.9707529558182949,
|
|
"step": 1560
|
|
},
|
|
{
|
|
"loss": 1.4377,
|
|
"grad_norm": 5.55445671081543,
|
|
"learning_rate": 3.1360524144810055e-07,
|
|
"epoch": 0.9769757311761046,
|
|
"step": 1570
|
|
},
|
|
{
|
|
"loss": 1.3551,
|
|
"grad_norm": 6.75304651260376,
|
|
"learning_rate": 1.703083714017617e-07,
|
|
"epoch": 0.9831985065339142,
|
|
"step": 1580
|
|
},
|
|
{
|
|
"loss": 1.4783,
|
|
"grad_norm": 5.284483432769775,
|
|
"learning_rate": 7.039426774164693e-08,
|
|
"epoch": 0.9894212818917237,
|
|
"step": 1590
|
|
},
|
|
{
|
|
"loss": 1.4281,
|
|
"grad_norm": 5.420683860778809,
|
|
"learning_rate": 1.3906349916881222e-08,
|
|
"epoch": 0.9956440572495333,
|
|
"step": 1600
|
|
},
|
|
{
|
|
"eval_loss": 1.4093519449234009,
|
|
"eval_runtime": 14.2389,
|
|
"eval_samples_per_second": 36.871,
|
|
"eval_steps_per_second": 9.27,
|
|
"epoch": 0.9956440572495333,
|
|
"step": 1600
|
|
},
|
|
{
|
|
"train_runtime": 1590.5658,
|
|
"train_samples_per_second": 16.162,
|
|
"train_steps_per_second": 1.01,
|
|
"total_flos": 1.172369019552e+16,
|
|
"train_loss": 1.5245253444830675,
|
|
"epoch": 1.0,
|
|
"step": 1607
|
|
}
|
|
] |