1535 lines
44 KiB
JSON
1535 lines
44 KiB
JSON
{
|
|
"epoch": 3.5724508050089447,
|
|
"global_step": 1000,
|
|
"max_steps": 2232,
|
|
"logging_steps": 5,
|
|
"eval_steps": 100,
|
|
"save_steps": 100,
|
|
"train_batch_size": 16,
|
|
"num_train_epochs": 8,
|
|
"num_input_tokens_seen": 0,
|
|
"total_flos": 4.722006162277663e+17,
|
|
"log_history": [
|
|
{
|
|
"loss": 10.1939,
|
|
"grad_norm": 74.8128433227539,
|
|
"learning_rate": 3.75e-05,
|
|
"epoch": 0.017889087656529516,
|
|
"step": 5
|
|
},
|
|
{
|
|
"loss": 3.9273,
|
|
"grad_norm": 12.053423881530762,
|
|
"learning_rate": 7.5e-05,
|
|
"epoch": 0.03577817531305903,
|
|
"step": 10
|
|
},
|
|
{
|
|
"loss": 1.4898,
|
|
"grad_norm": 40.130340576171875,
|
|
"learning_rate": 0.0001125,
|
|
"epoch": 0.05366726296958855,
|
|
"step": 15
|
|
},
|
|
{
|
|
"loss": 0.9064,
|
|
"grad_norm": 3.677988290786743,
|
|
"learning_rate": 0.00015,
|
|
"epoch": 0.07155635062611806,
|
|
"step": 20
|
|
},
|
|
{
|
|
"loss": 0.8129,
|
|
"grad_norm": 1.335487961769104,
|
|
"learning_rate": 0.00018749999999999998,
|
|
"epoch": 0.08944543828264759,
|
|
"step": 25
|
|
},
|
|
{
|
|
"loss": 1.1672,
|
|
"grad_norm": 29.67264175415039,
|
|
"learning_rate": 0.000225,
|
|
"epoch": 0.1073345259391771,
|
|
"step": 30
|
|
},
|
|
{
|
|
"loss": 0.7385,
|
|
"grad_norm": 1.039961576461792,
|
|
"learning_rate": 0.0002625,
|
|
"epoch": 0.1252236135957066,
|
|
"step": 35
|
|
},
|
|
{
|
|
"loss": 0.6884,
|
|
"grad_norm": 0.4314720928668976,
|
|
"learning_rate": 0.0003,
|
|
"epoch": 0.14311270125223613,
|
|
"step": 40
|
|
},
|
|
{
|
|
"loss": 0.6748,
|
|
"grad_norm": 0.40387117862701416,
|
|
"learning_rate": 0.0002999961486050259,
|
|
"epoch": 0.16100178890876565,
|
|
"step": 45
|
|
},
|
|
{
|
|
"loss": 0.6628,
|
|
"grad_norm": 0.448535293340683,
|
|
"learning_rate": 0.00029998459461788026,
|
|
"epoch": 0.17889087656529518,
|
|
"step": 50
|
|
},
|
|
{
|
|
"loss": 0.6659,
|
|
"grad_norm": 0.5291304588317871,
|
|
"learning_rate": 0.0002999653386318826,
|
|
"epoch": 0.1967799642218247,
|
|
"step": 55
|
|
},
|
|
{
|
|
"loss": 0.6517,
|
|
"grad_norm": 0.6947611570358276,
|
|
"learning_rate": 0.0002999383816358651,
|
|
"epoch": 0.2146690518783542,
|
|
"step": 60
|
|
},
|
|
{
|
|
"loss": 0.7263,
|
|
"grad_norm": 0.5037692785263062,
|
|
"learning_rate": 0.0002999037250141215,
|
|
"epoch": 0.23255813953488372,
|
|
"step": 65
|
|
},
|
|
{
|
|
"loss": 0.6464,
|
|
"grad_norm": 0.35925614833831787,
|
|
"learning_rate": 0.0002998613705463364,
|
|
"epoch": 0.2504472271914132,
|
|
"step": 70
|
|
},
|
|
{
|
|
"loss": 0.6531,
|
|
"grad_norm": 0.7011401653289795,
|
|
"learning_rate": 0.0002998113204074936,
|
|
"epoch": 0.26833631484794274,
|
|
"step": 75
|
|
},
|
|
{
|
|
"loss": 0.6563,
|
|
"grad_norm": 0.29620787501335144,
|
|
"learning_rate": 0.0002997535771677644,
|
|
"epoch": 0.28622540250447226,
|
|
"step": 80
|
|
},
|
|
{
|
|
"loss": 0.6555,
|
|
"grad_norm": 0.24026329815387726,
|
|
"learning_rate": 0.00029968814379237584,
|
|
"epoch": 0.3041144901610018,
|
|
"step": 85
|
|
},
|
|
{
|
|
"loss": 0.6213,
|
|
"grad_norm": 0.17133238911628723,
|
|
"learning_rate": 0.00029961502364145824,
|
|
"epoch": 0.3220035778175313,
|
|
"step": 90
|
|
},
|
|
{
|
|
"loss": 0.6216,
|
|
"grad_norm": 0.24976064264774323,
|
|
"learning_rate": 0.0002995342204698726,
|
|
"epoch": 0.33989266547406083,
|
|
"step": 95
|
|
},
|
|
{
|
|
"loss": 0.6357,
|
|
"grad_norm": 0.527025043964386,
|
|
"learning_rate": 0.00029944573842701806,
|
|
"epoch": 0.35778175313059035,
|
|
"step": 100
|
|
},
|
|
{
|
|
"eval_loss": 0.5987367630004883,
|
|
"eval_runtime": 20.6719,
|
|
"eval_samples_per_second": 162.83,
|
|
"eval_steps_per_second": 10.207,
|
|
"epoch": 0.35778175313059035,
|
|
"step": 100
|
|
},
|
|
{
|
|
"loss": 0.6205,
|
|
"grad_norm": 0.3432077467441559,
|
|
"learning_rate": 0.00029934958205661853,
|
|
"epoch": 0.3756708407871199,
|
|
"step": 105
|
|
},
|
|
{
|
|
"loss": 0.6071,
|
|
"grad_norm": 0.22968943417072296,
|
|
"learning_rate": 0.00029924575629648953,
|
|
"epoch": 0.3935599284436494,
|
|
"step": 110
|
|
},
|
|
{
|
|
"loss": 0.6096,
|
|
"grad_norm": 0.402422696352005,
|
|
"learning_rate": 0.0002991342664782845,
|
|
"epoch": 0.41144901610017887,
|
|
"step": 115
|
|
},
|
|
{
|
|
"loss": 0.597,
|
|
"grad_norm": 0.3375358283519745,
|
|
"learning_rate": 0.00029901511832722104,
|
|
"epoch": 0.4293381037567084,
|
|
"step": 120
|
|
},
|
|
{
|
|
"loss": 0.5821,
|
|
"grad_norm": 0.321890652179718,
|
|
"learning_rate": 0.00029888831796178714,
|
|
"epoch": 0.4472271914132379,
|
|
"step": 125
|
|
},
|
|
{
|
|
"loss": 0.5982,
|
|
"grad_norm": 0.20958741009235382,
|
|
"learning_rate": 0.0002987538718934266,
|
|
"epoch": 0.46511627906976744,
|
|
"step": 130
|
|
},
|
|
{
|
|
"loss": 0.5672,
|
|
"grad_norm": 0.2855433225631714,
|
|
"learning_rate": 0.000298611787026205,
|
|
"epoch": 0.48300536672629696,
|
|
"step": 135
|
|
},
|
|
{
|
|
"loss": 0.5756,
|
|
"grad_norm": 0.3264622688293457,
|
|
"learning_rate": 0.0002984620706564548,
|
|
"epoch": 0.5008944543828264,
|
|
"step": 140
|
|
},
|
|
{
|
|
"loss": 0.5643,
|
|
"grad_norm": 0.3137723505496979,
|
|
"learning_rate": 0.0002983047304724011,
|
|
"epoch": 0.518783542039356,
|
|
"step": 145
|
|
},
|
|
{
|
|
"loss": 0.5854,
|
|
"grad_norm": 0.2823493480682373,
|
|
"learning_rate": 0.00029813977455376634,
|
|
"epoch": 0.5366726296958855,
|
|
"step": 150
|
|
},
|
|
{
|
|
"loss": 0.5702,
|
|
"grad_norm": 0.24358761310577393,
|
|
"learning_rate": 0.00029796721137135597,
|
|
"epoch": 0.554561717352415,
|
|
"step": 155
|
|
},
|
|
{
|
|
"loss": 0.5487,
|
|
"grad_norm": 0.31340864300727844,
|
|
"learning_rate": 0.00029778704978662284,
|
|
"epoch": 0.5724508050089445,
|
|
"step": 160
|
|
},
|
|
{
|
|
"loss": 0.5478,
|
|
"grad_norm": 0.3177396357059479,
|
|
"learning_rate": 0.0002975992990512127,
|
|
"epoch": 0.590339892665474,
|
|
"step": 165
|
|
},
|
|
{
|
|
"loss": 0.5501,
|
|
"grad_norm": 0.3493712842464447,
|
|
"learning_rate": 0.0002974039688064886,
|
|
"epoch": 0.6082289803220036,
|
|
"step": 170
|
|
},
|
|
{
|
|
"loss": 0.5411,
|
|
"grad_norm": 0.2914203107357025,
|
|
"learning_rate": 0.00029720106908303625,
|
|
"epoch": 0.6261180679785331,
|
|
"step": 175
|
|
},
|
|
{
|
|
"loss": 0.5306,
|
|
"grad_norm": 0.29958376288414,
|
|
"learning_rate": 0.0002969906103001486,
|
|
"epoch": 0.6440071556350626,
|
|
"step": 180
|
|
},
|
|
{
|
|
"loss": 0.5304,
|
|
"grad_norm": 0.26977163553237915,
|
|
"learning_rate": 0.00029677260326529107,
|
|
"epoch": 0.6618962432915921,
|
|
"step": 185
|
|
},
|
|
{
|
|
"loss": 0.5231,
|
|
"grad_norm": 0.2629687786102295,
|
|
"learning_rate": 0.0002965470591735462,
|
|
"epoch": 0.6797853309481217,
|
|
"step": 190
|
|
},
|
|
{
|
|
"loss": 0.524,
|
|
"grad_norm": 0.3607335388660431,
|
|
"learning_rate": 0.0002963139896070391,
|
|
"epoch": 0.6976744186046512,
|
|
"step": 195
|
|
},
|
|
{
|
|
"loss": 0.522,
|
|
"grad_norm": 0.36577939987182617,
|
|
"learning_rate": 0.0002960734065343426,
|
|
"epoch": 0.7155635062611807,
|
|
"step": 200
|
|
},
|
|
{
|
|
"eval_loss": 0.5033101439476013,
|
|
"eval_runtime": 20.6749,
|
|
"eval_samples_per_second": 162.806,
|
|
"eval_steps_per_second": 10.206,
|
|
"epoch": 0.7155635062611807,
|
|
"step": 200
|
|
},
|
|
{
|
|
"loss": 0.5033,
|
|
"grad_norm": 0.30699774622917175,
|
|
"learning_rate": 0.00029582532230986244,
|
|
"epoch": 0.7334525939177102,
|
|
"step": 205
|
|
},
|
|
{
|
|
"loss": 0.5037,
|
|
"grad_norm": 0.27759912610054016,
|
|
"learning_rate": 0.0002955697496732031,
|
|
"epoch": 0.7513416815742398,
|
|
"step": 210
|
|
},
|
|
{
|
|
"loss": 0.5085,
|
|
"grad_norm": 0.32028377056121826,
|
|
"learning_rate": 0.0002953067017485136,
|
|
"epoch": 0.7692307692307693,
|
|
"step": 215
|
|
},
|
|
{
|
|
"loss": 0.493,
|
|
"grad_norm": 0.5791727304458618,
|
|
"learning_rate": 0.00029503619204381313,
|
|
"epoch": 0.7871198568872988,
|
|
"step": 220
|
|
},
|
|
{
|
|
"loss": 0.5089,
|
|
"grad_norm": 0.44022566080093384,
|
|
"learning_rate": 0.0002947582344502981,
|
|
"epoch": 0.8050089445438283,
|
|
"step": 225
|
|
},
|
|
{
|
|
"loss": 0.4986,
|
|
"grad_norm": 0.304227739572525,
|
|
"learning_rate": 0.0002944728432416282,
|
|
"epoch": 0.8228980322003577,
|
|
"step": 230
|
|
},
|
|
{
|
|
"loss": 0.4991,
|
|
"grad_norm": 0.2987827956676483,
|
|
"learning_rate": 0.0002941800330731936,
|
|
"epoch": 0.8407871198568873,
|
|
"step": 235
|
|
},
|
|
{
|
|
"loss": 0.4798,
|
|
"grad_norm": 0.32087433338165283,
|
|
"learning_rate": 0.0002938798189813625,
|
|
"epoch": 0.8586762075134168,
|
|
"step": 240
|
|
},
|
|
{
|
|
"loss": 0.4839,
|
|
"grad_norm": 0.354798287153244,
|
|
"learning_rate": 0.0002935722163827087,
|
|
"epoch": 0.8765652951699463,
|
|
"step": 245
|
|
},
|
|
{
|
|
"loss": 0.4874,
|
|
"grad_norm": 0.5133390426635742,
|
|
"learning_rate": 0.0002932572410732204,
|
|
"epoch": 0.8944543828264758,
|
|
"step": 250
|
|
},
|
|
{
|
|
"loss": 0.4749,
|
|
"grad_norm": 0.3183499574661255,
|
|
"learning_rate": 0.00029293490922748856,
|
|
"epoch": 0.9123434704830053,
|
|
"step": 255
|
|
},
|
|
{
|
|
"loss": 0.471,
|
|
"grad_norm": 0.3271007537841797,
|
|
"learning_rate": 0.0002926052373978764,
|
|
"epoch": 0.9302325581395349,
|
|
"step": 260
|
|
},
|
|
{
|
|
"loss": 0.459,
|
|
"grad_norm": 1.1660605669021606,
|
|
"learning_rate": 0.00029226824251366967,
|
|
"epoch": 0.9481216457960644,
|
|
"step": 265
|
|
},
|
|
{
|
|
"loss": 0.4369,
|
|
"grad_norm": 0.48370498418807983,
|
|
"learning_rate": 0.00029192394188020716,
|
|
"epoch": 0.9660107334525939,
|
|
"step": 270
|
|
},
|
|
{
|
|
"loss": 0.443,
|
|
"grad_norm": 0.3688034117221832,
|
|
"learning_rate": 0.0002915723531779919,
|
|
"epoch": 0.9838998211091234,
|
|
"step": 275
|
|
},
|
|
{
|
|
"loss": 0.4541,
|
|
"grad_norm": 0.7489350438117981,
|
|
"learning_rate": 0.00029121349446178333,
|
|
"epoch": 1.0,
|
|
"step": 280
|
|
},
|
|
{
|
|
"loss": 0.4349,
|
|
"grad_norm": 0.3898984491825104,
|
|
"learning_rate": 0.00029084738415967017,
|
|
"epoch": 1.0178890876565294,
|
|
"step": 285
|
|
},
|
|
{
|
|
"loss": 0.4455,
|
|
"grad_norm": 0.488726407289505,
|
|
"learning_rate": 0.0002904740410721242,
|
|
"epoch": 1.035778175313059,
|
|
"step": 290
|
|
},
|
|
{
|
|
"loss": 0.4431,
|
|
"grad_norm": 0.4695197641849518,
|
|
"learning_rate": 0.00029009348437103455,
|
|
"epoch": 1.0536672629695885,
|
|
"step": 295
|
|
},
|
|
{
|
|
"loss": 0.4223,
|
|
"grad_norm": 0.3967965543270111,
|
|
"learning_rate": 0.0002897057335987235,
|
|
"epoch": 1.071556350626118,
|
|
"step": 300
|
|
},
|
|
{
|
|
"eval_loss": 0.4258047640323639,
|
|
"eval_runtime": 20.6743,
|
|
"eval_samples_per_second": 162.811,
|
|
"eval_steps_per_second": 10.206,
|
|
"epoch": 1.071556350626118,
|
|
"step": 300
|
|
},
|
|
{
|
|
"loss": 0.4262,
|
|
"grad_norm": 0.502098023891449,
|
|
"learning_rate": 0.00028931080866694274,
|
|
"epoch": 1.0894454382826475,
|
|
"step": 305
|
|
},
|
|
{
|
|
"loss": 0.4155,
|
|
"grad_norm": 0.3177882134914398,
|
|
"learning_rate": 0.0002889087298558508,
|
|
"epoch": 1.1073345259391771,
|
|
"step": 310
|
|
},
|
|
{
|
|
"loss": 0.4151,
|
|
"grad_norm": 0.39488786458969116,
|
|
"learning_rate": 0.0002884995178129719,
|
|
"epoch": 1.1252236135957066,
|
|
"step": 315
|
|
},
|
|
{
|
|
"loss": 0.4025,
|
|
"grad_norm": 0.5109437704086304,
|
|
"learning_rate": 0.0002880831935521355,
|
|
"epoch": 1.1431127012522362,
|
|
"step": 320
|
|
},
|
|
{
|
|
"loss": 0.4054,
|
|
"grad_norm": 0.5653850436210632,
|
|
"learning_rate": 0.00028765977845239706,
|
|
"epoch": 1.1610017889087656,
|
|
"step": 325
|
|
},
|
|
{
|
|
"loss": 0.3982,
|
|
"grad_norm": 0.7002792358398438,
|
|
"learning_rate": 0.00028722929425694044,
|
|
"epoch": 1.1788908765652952,
|
|
"step": 330
|
|
},
|
|
{
|
|
"loss": 0.4375,
|
|
"grad_norm": 0.33770787715911865,
|
|
"learning_rate": 0.0002867917630719612,
|
|
"epoch": 1.1967799642218246,
|
|
"step": 335
|
|
},
|
|
{
|
|
"loss": 0.4053,
|
|
"grad_norm": 0.4120868444442749,
|
|
"learning_rate": 0.00028634720736553143,
|
|
"epoch": 1.2146690518783543,
|
|
"step": 340
|
|
},
|
|
{
|
|
"loss": 0.411,
|
|
"grad_norm": 0.4081322252750397,
|
|
"learning_rate": 0.00028589564996644594,
|
|
"epoch": 1.2325581395348837,
|
|
"step": 345
|
|
},
|
|
{
|
|
"loss": 0.4167,
|
|
"grad_norm": 0.6889946460723877,
|
|
"learning_rate": 0.0002854371140630501,
|
|
"epoch": 1.250447227191413,
|
|
"step": 350
|
|
},
|
|
{
|
|
"loss": 0.4196,
|
|
"grad_norm": 0.3247791826725006,
|
|
"learning_rate": 0.00028497162320204885,
|
|
"epoch": 1.2683363148479427,
|
|
"step": 355
|
|
},
|
|
{
|
|
"loss": 0.4091,
|
|
"grad_norm": 0.41395092010498047,
|
|
"learning_rate": 0.00028449920128729766,
|
|
"epoch": 1.2862254025044724,
|
|
"step": 360
|
|
},
|
|
{
|
|
"loss": 0.3958,
|
|
"grad_norm": 0.3653516173362732,
|
|
"learning_rate": 0.00028401987257857514,
|
|
"epoch": 1.3041144901610018,
|
|
"step": 365
|
|
},
|
|
{
|
|
"loss": 0.3898,
|
|
"grad_norm": 0.3595719337463379,
|
|
"learning_rate": 0.0002835336616903369,
|
|
"epoch": 1.3220035778175312,
|
|
"step": 370
|
|
},
|
|
{
|
|
"loss": 0.3604,
|
|
"grad_norm": 0.7938292622566223,
|
|
"learning_rate": 0.000283040593590452,
|
|
"epoch": 1.3398926654740608,
|
|
"step": 375
|
|
},
|
|
{
|
|
"loss": 0.382,
|
|
"grad_norm": 0.562586784362793,
|
|
"learning_rate": 0.00028254069359892034,
|
|
"epoch": 1.3577817531305905,
|
|
"step": 380
|
|
},
|
|
{
|
|
"loss": 0.385,
|
|
"grad_norm": 0.29362237453460693,
|
|
"learning_rate": 0.0002820339873865729,
|
|
"epoch": 1.3756708407871199,
|
|
"step": 385
|
|
},
|
|
{
|
|
"loss": 0.3776,
|
|
"grad_norm": 0.38946282863616943,
|
|
"learning_rate": 0.00028152050097375304,
|
|
"epoch": 1.3935599284436493,
|
|
"step": 390
|
|
},
|
|
{
|
|
"loss": 0.3756,
|
|
"grad_norm": 0.3675141930580139,
|
|
"learning_rate": 0.00028100026072898067,
|
|
"epoch": 1.411449016100179,
|
|
"step": 395
|
|
},
|
|
{
|
|
"loss": 0.3627,
|
|
"grad_norm": 0.4281761050224304,
|
|
"learning_rate": 0.00028047329336759806,
|
|
"epoch": 1.4293381037567083,
|
|
"step": 400
|
|
},
|
|
{
|
|
"eval_loss": 0.3722744286060333,
|
|
"eval_runtime": 20.6775,
|
|
"eval_samples_per_second": 162.785,
|
|
"eval_steps_per_second": 10.204,
|
|
"epoch": 1.4293381037567083,
|
|
"step": 400
|
|
},
|
|
{
|
|
"loss": 0.3637,
|
|
"grad_norm": 0.3468483090400696,
|
|
"learning_rate": 0.0002799396259503977,
|
|
"epoch": 1.447227191413238,
|
|
"step": 405
|
|
},
|
|
{
|
|
"loss": 0.3729,
|
|
"grad_norm": 0.4413630962371826,
|
|
"learning_rate": 0.00027939928588223314,
|
|
"epoch": 1.4651162790697674,
|
|
"step": 410
|
|
},
|
|
{
|
|
"loss": 0.3511,
|
|
"grad_norm": 0.41624078154563904,
|
|
"learning_rate": 0.00027885230091061127,
|
|
"epoch": 1.483005366726297,
|
|
"step": 415
|
|
},
|
|
{
|
|
"loss": 0.3599,
|
|
"grad_norm": 0.42356589436531067,
|
|
"learning_rate": 0.00027829869912426777,
|
|
"epoch": 1.5008944543828264,
|
|
"step": 420
|
|
},
|
|
{
|
|
"loss": 0.3586,
|
|
"grad_norm": 0.4348715543746948,
|
|
"learning_rate": 0.0002777385089517244,
|
|
"epoch": 1.518783542039356,
|
|
"step": 425
|
|
},
|
|
{
|
|
"loss": 0.3487,
|
|
"grad_norm": 0.5163382291793823,
|
|
"learning_rate": 0.00027717175915982945,
|
|
"epoch": 1.5366726296958855,
|
|
"step": 430
|
|
},
|
|
{
|
|
"loss": 0.3551,
|
|
"grad_norm": 0.47645968198776245,
|
|
"learning_rate": 0.0002765984788522801,
|
|
"epoch": 1.5545617173524149,
|
|
"step": 435
|
|
},
|
|
{
|
|
"loss": 0.3523,
|
|
"grad_norm": 0.5956697463989258,
|
|
"learning_rate": 0.0002760186974681285,
|
|
"epoch": 1.5724508050089445,
|
|
"step": 440
|
|
},
|
|
{
|
|
"loss": 0.3842,
|
|
"grad_norm": 0.3199741244316101,
|
|
"learning_rate": 0.0002754324447802693,
|
|
"epoch": 1.5903398926654742,
|
|
"step": 445
|
|
},
|
|
{
|
|
"loss": 0.363,
|
|
"grad_norm": 0.4985601007938385,
|
|
"learning_rate": 0.00027483975089391126,
|
|
"epoch": 1.6082289803220036,
|
|
"step": 450
|
|
},
|
|
{
|
|
"loss": 0.3573,
|
|
"grad_norm": 0.43190059065818787,
|
|
"learning_rate": 0.0002742406462450311,
|
|
"epoch": 1.626118067978533,
|
|
"step": 455
|
|
},
|
|
{
|
|
"loss": 0.3399,
|
|
"grad_norm": 0.3566244840621948,
|
|
"learning_rate": 0.00027363516159881066,
|
|
"epoch": 1.6440071556350626,
|
|
"step": 460
|
|
},
|
|
{
|
|
"loss": 0.3465,
|
|
"grad_norm": 0.4065592586994171,
|
|
"learning_rate": 0.0002730233280480569,
|
|
"epoch": 1.6618962432915922,
|
|
"step": 465
|
|
},
|
|
{
|
|
"loss": 0.3324,
|
|
"grad_norm": 0.36722782254219055,
|
|
"learning_rate": 0.0002724051770116052,
|
|
"epoch": 1.6797853309481217,
|
|
"step": 470
|
|
},
|
|
{
|
|
"loss": 0.325,
|
|
"grad_norm": 0.3565702736377716,
|
|
"learning_rate": 0.00027178074023270635,
|
|
"epoch": 1.697674418604651,
|
|
"step": 475
|
|
},
|
|
{
|
|
"loss": 0.3679,
|
|
"grad_norm": 0.5317059755325317,
|
|
"learning_rate": 0.00027115004977739586,
|
|
"epoch": 1.7155635062611807,
|
|
"step": 480
|
|
},
|
|
{
|
|
"loss": 0.3433,
|
|
"grad_norm": 0.30274462699890137,
|
|
"learning_rate": 0.00027051313803284774,
|
|
"epoch": 1.7334525939177103,
|
|
"step": 485
|
|
},
|
|
{
|
|
"loss": 0.3492,
|
|
"grad_norm": 4.536398887634277,
|
|
"learning_rate": 0.00026987003770571124,
|
|
"epoch": 1.7513416815742398,
|
|
"step": 490
|
|
},
|
|
{
|
|
"loss": 0.3506,
|
|
"grad_norm": 0.4281180202960968,
|
|
"learning_rate": 0.0002692207818204313,
|
|
"epoch": 1.7692307692307692,
|
|
"step": 495
|
|
},
|
|
{
|
|
"loss": 0.3417,
|
|
"grad_norm": 0.3851430416107178,
|
|
"learning_rate": 0.0002685654037175525,
|
|
"epoch": 1.7871198568872988,
|
|
"step": 500
|
|
},
|
|
{
|
|
"eval_loss": 0.3463773727416992,
|
|
"eval_runtime": 20.6737,
|
|
"eval_samples_per_second": 162.815,
|
|
"eval_steps_per_second": 10.206,
|
|
"epoch": 1.7871198568872988,
|
|
"step": 500
|
|
},
|
|
{
|
|
"loss": 0.3287,
|
|
"grad_norm": 0.41302162408828735,
|
|
"learning_rate": 0.0002679039370520074,
|
|
"epoch": 1.8050089445438284,
|
|
"step": 505
|
|
},
|
|
{
|
|
"loss": 0.3292,
|
|
"grad_norm": 0.3512372076511383,
|
|
"learning_rate": 0.00026723641579138777,
|
|
"epoch": 1.8228980322003578,
|
|
"step": 510
|
|
},
|
|
{
|
|
"loss": 0.3241,
|
|
"grad_norm": 0.4498589336872101,
|
|
"learning_rate": 0.0002665628742142008,
|
|
"epoch": 1.8407871198568873,
|
|
"step": 515
|
|
},
|
|
{
|
|
"loss": 0.3302,
|
|
"grad_norm": 0.532430112361908,
|
|
"learning_rate": 0.0002658833469081082,
|
|
"epoch": 1.8586762075134167,
|
|
"step": 520
|
|
},
|
|
{
|
|
"loss": 0.3254,
|
|
"grad_norm": 0.40190622210502625,
|
|
"learning_rate": 0.0002651978687681509,
|
|
"epoch": 1.8765652951699463,
|
|
"step": 525
|
|
},
|
|
{
|
|
"loss": 0.3212,
|
|
"grad_norm": 0.4508027732372284,
|
|
"learning_rate": 0.00026450647499495626,
|
|
"epoch": 1.894454382826476,
|
|
"step": 530
|
|
},
|
|
{
|
|
"loss": 0.3191,
|
|
"grad_norm": 0.48150140047073364,
|
|
"learning_rate": 0.00026380920109293104,
|
|
"epoch": 1.9123434704830053,
|
|
"step": 535
|
|
},
|
|
{
|
|
"loss": 0.3374,
|
|
"grad_norm": 0.34987515211105347,
|
|
"learning_rate": 0.00026310608286843795,
|
|
"epoch": 1.9302325581395348,
|
|
"step": 540
|
|
},
|
|
{
|
|
"loss": 0.3193,
|
|
"grad_norm": 0.2754597067832947,
|
|
"learning_rate": 0.00026239715642795684,
|
|
"epoch": 1.9481216457960644,
|
|
"step": 545
|
|
},
|
|
{
|
|
"loss": 0.3276,
|
|
"grad_norm": 0.35444873571395874,
|
|
"learning_rate": 0.00026168245817623085,
|
|
"epoch": 1.966010733452594,
|
|
"step": 550
|
|
},
|
|
{
|
|
"loss": 0.3028,
|
|
"grad_norm": 0.3781956434249878,
|
|
"learning_rate": 0.00026096202481439674,
|
|
"epoch": 1.9838998211091234,
|
|
"step": 555
|
|
},
|
|
{
|
|
"loss": 0.3163,
|
|
"grad_norm": 0.43553584814071655,
|
|
"learning_rate": 0.00026023589333810015,
|
|
"epoch": 2.0,
|
|
"step": 560
|
|
},
|
|
{
|
|
"loss": 0.3041,
|
|
"grad_norm": 0.3745397627353668,
|
|
"learning_rate": 0.00025950410103559604,
|
|
"epoch": 2.0178890876565294,
|
|
"step": 565
|
|
},
|
|
{
|
|
"loss": 0.314,
|
|
"grad_norm": 0.4507661759853363,
|
|
"learning_rate": 0.00025876668548583374,
|
|
"epoch": 2.035778175313059,
|
|
"step": 570
|
|
},
|
|
{
|
|
"loss": 0.2911,
|
|
"grad_norm": 0.45311978459358215,
|
|
"learning_rate": 0.00025802368455652704,
|
|
"epoch": 2.0536672629695887,
|
|
"step": 575
|
|
},
|
|
{
|
|
"loss": 0.2791,
|
|
"grad_norm": 0.3784424066543579,
|
|
"learning_rate": 0.00025727513640220985,
|
|
"epoch": 2.071556350626118,
|
|
"step": 580
|
|
},
|
|
{
|
|
"loss": 0.2781,
|
|
"grad_norm": 0.37560757994651794,
|
|
"learning_rate": 0.000256521079462277,
|
|
"epoch": 2.0894454382826475,
|
|
"step": 585
|
|
},
|
|
{
|
|
"loss": 0.2705,
|
|
"grad_norm": 0.4536455571651459,
|
|
"learning_rate": 0.0002557615524590097,
|
|
"epoch": 2.107334525939177,
|
|
"step": 590
|
|
},
|
|
{
|
|
"loss": 0.2845,
|
|
"grad_norm": 0.8856727480888367,
|
|
"learning_rate": 0.000254996594395588,
|
|
"epoch": 2.1252236135957068,
|
|
"step": 595
|
|
},
|
|
{
|
|
"loss": 0.2894,
|
|
"grad_norm": 0.5242048501968384,
|
|
"learning_rate": 0.00025422624455408686,
|
|
"epoch": 2.143112701252236,
|
|
"step": 600
|
|
},
|
|
{
|
|
"eval_loss": 0.33662545680999756,
|
|
"eval_runtime": 20.6734,
|
|
"eval_samples_per_second": 162.818,
|
|
"eval_steps_per_second": 10.206,
|
|
"epoch": 2.143112701252236,
|
|
"step": 600
|
|
},
|
|
{
|
|
"loss": 0.2955,
|
|
"grad_norm": 0.37391841411590576,
|
|
"learning_rate": 0.0002534505424934599,
|
|
"epoch": 2.1610017889087656,
|
|
"step": 605
|
|
},
|
|
{
|
|
"loss": 0.2869,
|
|
"grad_norm": 0.3889912962913513,
|
|
"learning_rate": 0.00025266952804750727,
|
|
"epoch": 2.178890876565295,
|
|
"step": 610
|
|
},
|
|
{
|
|
"loss": 0.2805,
|
|
"grad_norm": 0.37954917550086975,
|
|
"learning_rate": 0.0002518832413228304,
|
|
"epoch": 2.196779964221825,
|
|
"step": 615
|
|
},
|
|
{
|
|
"loss": 0.2834,
|
|
"grad_norm": 0.3385255038738251,
|
|
"learning_rate": 0.00025109172269677265,
|
|
"epoch": 2.2146690518783543,
|
|
"step": 620
|
|
},
|
|
{
|
|
"loss": 0.2741,
|
|
"grad_norm": 0.33614033460617065,
|
|
"learning_rate": 0.00025029501281534534,
|
|
"epoch": 2.2325581395348837,
|
|
"step": 625
|
|
},
|
|
{
|
|
"loss": 0.2803,
|
|
"grad_norm": 0.4882681369781494,
|
|
"learning_rate": 0.00024949315259114094,
|
|
"epoch": 2.250447227191413,
|
|
"step": 630
|
|
},
|
|
{
|
|
"loss": 0.305,
|
|
"grad_norm": 0.39836975932121277,
|
|
"learning_rate": 0.00024868618320123193,
|
|
"epoch": 2.268336314847943,
|
|
"step": 635
|
|
},
|
|
{
|
|
"loss": 0.2794,
|
|
"grad_norm": 0.30090266466140747,
|
|
"learning_rate": 0.00024787414608505636,
|
|
"epoch": 2.2862254025044724,
|
|
"step": 640
|
|
},
|
|
{
|
|
"loss": 0.2923,
|
|
"grad_norm": 0.40115466713905334,
|
|
"learning_rate": 0.0002470570829422898,
|
|
"epoch": 2.304114490161002,
|
|
"step": 645
|
|
},
|
|
{
|
|
"loss": 0.2752,
|
|
"grad_norm": 0.35384082794189453,
|
|
"learning_rate": 0.00024623503573070407,
|
|
"epoch": 2.322003577817531,
|
|
"step": 650
|
|
},
|
|
{
|
|
"loss": 0.2647,
|
|
"grad_norm": 0.5324997901916504,
|
|
"learning_rate": 0.00024540804666401235,
|
|
"epoch": 2.3398926654740606,
|
|
"step": 655
|
|
},
|
|
{
|
|
"loss": 0.2667,
|
|
"grad_norm": 0.3377424478530884,
|
|
"learning_rate": 0.00024457615820970194,
|
|
"epoch": 2.3577817531305905,
|
|
"step": 660
|
|
},
|
|
{
|
|
"loss": 0.273,
|
|
"grad_norm": 0.3497815728187561,
|
|
"learning_rate": 0.0002437394130868529,
|
|
"epoch": 2.37567084078712,
|
|
"step": 665
|
|
},
|
|
{
|
|
"loss": 0.2674,
|
|
"grad_norm": 0.4024289548397064,
|
|
"learning_rate": 0.00024289785426394471,
|
|
"epoch": 2.3935599284436493,
|
|
"step": 670
|
|
},
|
|
{
|
|
"loss": 0.2552,
|
|
"grad_norm": 0.2747831642627716,
|
|
"learning_rate": 0.00024205152495664965,
|
|
"epoch": 2.4114490161001787,
|
|
"step": 675
|
|
},
|
|
{
|
|
"loss": 0.2775,
|
|
"grad_norm": 0.5090282559394836,
|
|
"learning_rate": 0.00024120046862561364,
|
|
"epoch": 2.4293381037567086,
|
|
"step": 680
|
|
},
|
|
{
|
|
"loss": 0.2638,
|
|
"grad_norm": 0.423360675573349,
|
|
"learning_rate": 0.0002403447289742243,
|
|
"epoch": 2.447227191413238,
|
|
"step": 685
|
|
},
|
|
{
|
|
"loss": 0.2629,
|
|
"grad_norm": 0.3922378122806549,
|
|
"learning_rate": 0.0002394843499463669,
|
|
"epoch": 2.4651162790697674,
|
|
"step": 690
|
|
},
|
|
{
|
|
"loss": 0.2642,
|
|
"grad_norm": 0.381626158952713,
|
|
"learning_rate": 0.00023861937572416757,
|
|
"epoch": 2.483005366726297,
|
|
"step": 695
|
|
},
|
|
{
|
|
"loss": 0.2741,
|
|
"grad_norm": 0.41004201769828796,
|
|
"learning_rate": 0.0002377498507257247,
|
|
"epoch": 2.500894454382826,
|
|
"step": 700
|
|
},
|
|
{
|
|
"eval_loss": 0.3169912099838257,
|
|
"eval_runtime": 20.6812,
|
|
"eval_samples_per_second": 162.756,
|
|
"eval_steps_per_second": 10.202,
|
|
"epoch": 2.500894454382826,
|
|
"step": 700
|
|
},
|
|
{
|
|
"loss": 0.2669,
|
|
"grad_norm": 0.36008498072624207,
|
|
"learning_rate": 0.00023687581960282763,
|
|
"epoch": 2.518783542039356,
|
|
"step": 705
|
|
},
|
|
{
|
|
"loss": 0.2529,
|
|
"grad_norm": 0.47046661376953125,
|
|
"learning_rate": 0.00023599732723866414,
|
|
"epoch": 2.5366726296958855,
|
|
"step": 710
|
|
},
|
|
{
|
|
"loss": 0.2882,
|
|
"grad_norm": 0.6661579012870789,
|
|
"learning_rate": 0.00023511441874551512,
|
|
"epoch": 2.554561717352415,
|
|
"step": 715
|
|
},
|
|
{
|
|
"loss": 0.2805,
|
|
"grad_norm": 0.3578526973724365,
|
|
"learning_rate": 0.00023422713946243842,
|
|
"epoch": 2.5724508050089447,
|
|
"step": 720
|
|
},
|
|
{
|
|
"loss": 0.2615,
|
|
"grad_norm": 0.42732375860214233,
|
|
"learning_rate": 0.0002333355349529403,
|
|
"epoch": 2.590339892665474,
|
|
"step": 725
|
|
},
|
|
{
|
|
"loss": 0.2635,
|
|
"grad_norm": 0.3077392876148224,
|
|
"learning_rate": 0.0002324396510026358,
|
|
"epoch": 2.6082289803220036,
|
|
"step": 730
|
|
},
|
|
{
|
|
"loss": 0.2575,
|
|
"grad_norm": 0.3254878520965576,
|
|
"learning_rate": 0.00023153953361689753,
|
|
"epoch": 2.626118067978533,
|
|
"step": 735
|
|
},
|
|
{
|
|
"loss": 0.2539,
|
|
"grad_norm": 0.3971039056777954,
|
|
"learning_rate": 0.00023063522901849303,
|
|
"epoch": 2.6440071556350624,
|
|
"step": 740
|
|
},
|
|
{
|
|
"loss": 0.2675,
|
|
"grad_norm": 0.36017006635665894,
|
|
"learning_rate": 0.00022972678364521155,
|
|
"epoch": 2.6618962432915922,
|
|
"step": 745
|
|
},
|
|
{
|
|
"loss": 0.2752,
|
|
"grad_norm": 0.3148644268512726,
|
|
"learning_rate": 0.00022881424414747902,
|
|
"epoch": 2.6797853309481217,
|
|
"step": 750
|
|
},
|
|
{
|
|
"loss": 0.257,
|
|
"grad_norm": 0.46191850304603577,
|
|
"learning_rate": 0.00022789765738596258,
|
|
"epoch": 2.697674418604651,
|
|
"step": 755
|
|
},
|
|
{
|
|
"loss": 0.2602,
|
|
"grad_norm": 0.5142350792884827,
|
|
"learning_rate": 0.00022697707042916413,
|
|
"epoch": 2.715563506261181,
|
|
"step": 760
|
|
},
|
|
{
|
|
"loss": 0.2619,
|
|
"grad_norm": 0.4233485162258148,
|
|
"learning_rate": 0.00022605253055100346,
|
|
"epoch": 2.7334525939177103,
|
|
"step": 765
|
|
},
|
|
{
|
|
"loss": 0.2699,
|
|
"grad_norm": 0.36081498861312866,
|
|
"learning_rate": 0.00022512408522839034,
|
|
"epoch": 2.7513416815742398,
|
|
"step": 770
|
|
},
|
|
{
|
|
"loss": 0.2589,
|
|
"grad_norm": 0.3283044695854187,
|
|
"learning_rate": 0.0002241917821387868,
|
|
"epoch": 2.769230769230769,
|
|
"step": 775
|
|
},
|
|
{
|
|
"loss": 0.2569,
|
|
"grad_norm": 0.36509019136428833,
|
|
"learning_rate": 0.00022325566915775872,
|
|
"epoch": 2.7871198568872986,
|
|
"step": 780
|
|
},
|
|
{
|
|
"loss": 0.2631,
|
|
"grad_norm": 0.30786919593811035,
|
|
"learning_rate": 0.000222315794356517,
|
|
"epoch": 2.8050089445438284,
|
|
"step": 785
|
|
},
|
|
{
|
|
"loss": 0.2596,
|
|
"grad_norm": 0.26746758818626404,
|
|
"learning_rate": 0.0002213722059994496,
|
|
"epoch": 2.822898032200358,
|
|
"step": 790
|
|
},
|
|
{
|
|
"loss": 0.2462,
|
|
"grad_norm": 0.3422580659389496,
|
|
"learning_rate": 0.00022042495254164248,
|
|
"epoch": 2.8407871198568873,
|
|
"step": 795
|
|
},
|
|
{
|
|
"loss": 0.2604,
|
|
"grad_norm": 0.3908096253871918,
|
|
"learning_rate": 0.0002194740826263918,
|
|
"epoch": 2.8586762075134167,
|
|
"step": 800
|
|
},
|
|
{
|
|
"eval_loss": 0.31991705298423767,
|
|
"eval_runtime": 20.6861,
|
|
"eval_samples_per_second": 162.718,
|
|
"eval_steps_per_second": 10.2,
|
|
"epoch": 2.8586762075134167,
|
|
"step": 800
|
|
},
|
|
{
|
|
"loss": 0.259,
|
|
"grad_norm": 0.3868574798107147,
|
|
"learning_rate": 0.00021851964508270573,
|
|
"epoch": 2.8765652951699465,
|
|
"step": 805
|
|
},
|
|
{
|
|
"loss": 0.2627,
|
|
"grad_norm": 0.3655846416950226,
|
|
"learning_rate": 0.00021756168892279705,
|
|
"epoch": 2.894454382826476,
|
|
"step": 810
|
|
},
|
|
{
|
|
"loss": 0.249,
|
|
"grad_norm": 0.6286638379096985,
|
|
"learning_rate": 0.00021660026333956625,
|
|
"epoch": 2.9123434704830053,
|
|
"step": 815
|
|
},
|
|
{
|
|
"loss": 0.2517,
|
|
"grad_norm": 0.5203955769538879,
|
|
"learning_rate": 0.0002156354177040755,
|
|
"epoch": 2.9302325581395348,
|
|
"step": 820
|
|
},
|
|
{
|
|
"loss": 0.2665,
|
|
"grad_norm": 0.38546180725097656,
|
|
"learning_rate": 0.00021466720156301314,
|
|
"epoch": 2.948121645796064,
|
|
"step": 825
|
|
},
|
|
{
|
|
"loss": 0.2651,
|
|
"grad_norm": 0.3447847366333008,
|
|
"learning_rate": 0.00021369566463614968,
|
|
"epoch": 2.966010733452594,
|
|
"step": 830
|
|
},
|
|
{
|
|
"loss": 0.2494,
|
|
"grad_norm": 0.6365504264831543,
|
|
"learning_rate": 0.00021272085681378418,
|
|
"epoch": 2.9838998211091234,
|
|
"step": 835
|
|
},
|
|
{
|
|
"loss": 0.2448,
|
|
"grad_norm": 0.6940473914146423,
|
|
"learning_rate": 0.0002117428281541827,
|
|
"epoch": 3.0,
|
|
"step": 840
|
|
},
|
|
{
|
|
"loss": 0.1991,
|
|
"grad_norm": 0.3761572539806366,
|
|
"learning_rate": 0.00021076162888100734,
|
|
"epoch": 3.0178890876565294,
|
|
"step": 845
|
|
},
|
|
{
|
|
"loss": 0.1976,
|
|
"grad_norm": 0.3070697784423828,
|
|
"learning_rate": 0.00020977730938073747,
|
|
"epoch": 3.035778175313059,
|
|
"step": 850
|
|
},
|
|
{
|
|
"loss": 0.188,
|
|
"grad_norm": 0.5102139115333557,
|
|
"learning_rate": 0.0002087899202000821,
|
|
"epoch": 3.0536672629695887,
|
|
"step": 855
|
|
},
|
|
{
|
|
"loss": 0.2003,
|
|
"grad_norm": 0.4111884534358978,
|
|
"learning_rate": 0.00020779951204338422,
|
|
"epoch": 3.071556350626118,
|
|
"step": 860
|
|
},
|
|
{
|
|
"loss": 0.1914,
|
|
"grad_norm": 0.46625199913978577,
|
|
"learning_rate": 0.00020680613577001724,
|
|
"epoch": 3.0894454382826475,
|
|
"step": 865
|
|
},
|
|
{
|
|
"loss": 0.2037,
|
|
"grad_norm": 0.4094313383102417,
|
|
"learning_rate": 0.0002058098423917729,
|
|
"epoch": 3.107334525939177,
|
|
"step": 870
|
|
},
|
|
{
|
|
"loss": 0.2,
|
|
"grad_norm": 0.4287453591823578,
|
|
"learning_rate": 0.000204810683070242,
|
|
"epoch": 3.1252236135957068,
|
|
"step": 875
|
|
},
|
|
{
|
|
"loss": 0.1918,
|
|
"grad_norm": 0.4345555901527405,
|
|
"learning_rate": 0.00020380870911418709,
|
|
"epoch": 3.143112701252236,
|
|
"step": 880
|
|
},
|
|
{
|
|
"loss": 0.1911,
|
|
"grad_norm": 0.43231672048568726,
|
|
"learning_rate": 0.00020280397197690753,
|
|
"epoch": 3.1610017889087656,
|
|
"step": 885
|
|
},
|
|
{
|
|
"loss": 0.196,
|
|
"grad_norm": 0.4027317464351654,
|
|
"learning_rate": 0.00020179652325359756,
|
|
"epoch": 3.178890876565295,
|
|
"step": 890
|
|
},
|
|
{
|
|
"loss": 0.1964,
|
|
"grad_norm": 0.4894247353076935,
|
|
"learning_rate": 0.00020078641467869652,
|
|
"epoch": 3.196779964221825,
|
|
"step": 895
|
|
},
|
|
{
|
|
"loss": 0.1847,
|
|
"grad_norm": 0.4106491804122925,
|
|
"learning_rate": 0.00019977369812323214,
|
|
"epoch": 3.2146690518783543,
|
|
"step": 900
|
|
},
|
|
{
|
|
"eval_loss": 0.3646461069583893,
|
|
"eval_runtime": 20.6747,
|
|
"eval_samples_per_second": 162.808,
|
|
"eval_steps_per_second": 10.206,
|
|
"epoch": 3.2146690518783543,
|
|
"step": 900
|
|
},
|
|
{
|
|
"loss": 0.1918,
|
|
"grad_norm": 0.5217190980911255,
|
|
"learning_rate": 0.00019875842559215724,
|
|
"epoch": 3.2325581395348837,
|
|
"step": 905
|
|
},
|
|
{
|
|
"loss": 0.1856,
|
|
"grad_norm": 0.3551487922668457,
|
|
"learning_rate": 0.00019774064922167876,
|
|
"epoch": 3.250447227191413,
|
|
"step": 910
|
|
},
|
|
{
|
|
"loss": 0.1978,
|
|
"grad_norm": 0.44846031069755554,
|
|
"learning_rate": 0.00019672042127658062,
|
|
"epoch": 3.268336314847943,
|
|
"step": 915
|
|
},
|
|
{
|
|
"loss": 0.1912,
|
|
"grad_norm": 0.45646387338638306,
|
|
"learning_rate": 0.00019569779414753999,
|
|
"epoch": 3.2862254025044724,
|
|
"step": 920
|
|
},
|
|
{
|
|
"loss": 0.1988,
|
|
"grad_norm": 0.42022913694381714,
|
|
"learning_rate": 0.00019467282034843654,
|
|
"epoch": 3.304114490161002,
|
|
"step": 925
|
|
},
|
|
{
|
|
"loss": 0.188,
|
|
"grad_norm": 0.44034436345100403,
|
|
"learning_rate": 0.00019364555251365627,
|
|
"epoch": 3.322003577817531,
|
|
"step": 930
|
|
},
|
|
{
|
|
"loss": 0.2039,
|
|
"grad_norm": 0.5154224634170532,
|
|
"learning_rate": 0.00019261604339538809,
|
|
"epoch": 3.3398926654740606,
|
|
"step": 935
|
|
},
|
|
{
|
|
"loss": 0.2013,
|
|
"grad_norm": 0.6147205829620361,
|
|
"learning_rate": 0.0001915843458609152,
|
|
"epoch": 3.3577817531305905,
|
|
"step": 940
|
|
},
|
|
{
|
|
"loss": 0.2017,
|
|
"grad_norm": 0.460700660943985,
|
|
"learning_rate": 0.0001905505128899004,
|
|
"epoch": 3.37567084078712,
|
|
"step": 945
|
|
},
|
|
{
|
|
"loss": 0.1908,
|
|
"grad_norm": 0.5751897692680359,
|
|
"learning_rate": 0.000189514597571665,
|
|
"epoch": 3.3935599284436493,
|
|
"step": 950
|
|
},
|
|
{
|
|
"loss": 0.1965,
|
|
"grad_norm": 0.4614831507205963,
|
|
"learning_rate": 0.00018847665310246312,
|
|
"epoch": 3.4114490161001787,
|
|
"step": 955
|
|
},
|
|
{
|
|
"loss": 0.2014,
|
|
"grad_norm": 0.3863570988178253,
|
|
"learning_rate": 0.00018743673278274954,
|
|
"epoch": 3.4293381037567086,
|
|
"step": 960
|
|
},
|
|
{
|
|
"loss": 0.189,
|
|
"grad_norm": 0.3508825898170471,
|
|
"learning_rate": 0.0001863948900144428,
|
|
"epoch": 3.447227191413238,
|
|
"step": 965
|
|
},
|
|
{
|
|
"loss": 0.1793,
|
|
"grad_norm": 0.3714524805545807,
|
|
"learning_rate": 0.00018535117829818293,
|
|
"epoch": 3.4651162790697674,
|
|
"step": 970
|
|
},
|
|
{
|
|
"loss": 0.2049,
|
|
"grad_norm": 0.43370577692985535,
|
|
"learning_rate": 0.00018430565123058409,
|
|
"epoch": 3.483005366726297,
|
|
"step": 975
|
|
},
|
|
{
|
|
"loss": 0.1803,
|
|
"grad_norm": 0.34576115012168884,
|
|
"learning_rate": 0.00018325836250148215,
|
|
"epoch": 3.500894454382826,
|
|
"step": 980
|
|
},
|
|
{
|
|
"loss": 0.1891,
|
|
"grad_norm": 0.41390174627304077,
|
|
"learning_rate": 0.00018220936589117775,
|
|
"epoch": 3.518783542039356,
|
|
"step": 985
|
|
},
|
|
{
|
|
"loss": 0.1909,
|
|
"grad_norm": 0.5129839777946472,
|
|
"learning_rate": 0.00018115871526767455,
|
|
"epoch": 3.5366726296958855,
|
|
"step": 990
|
|
},
|
|
{
|
|
"loss": 0.1919,
|
|
"grad_norm": 0.537712812423706,
|
|
"learning_rate": 0.00018010646458391293,
|
|
"epoch": 3.554561717352415,
|
|
"step": 995
|
|
},
|
|
{
|
|
"loss": 0.1909,
|
|
"grad_norm": 0.3729874789714813,
|
|
"learning_rate": 0.00017905266787499947,
|
|
"epoch": 3.5724508050089447,
|
|
"step": 1000
|
|
},
|
|
{
|
|
"eval_loss": 0.36307844519615173,
|
|
"eval_runtime": 20.6777,
|
|
"eval_samples_per_second": 162.784,
|
|
"eval_steps_per_second": 10.204,
|
|
"epoch": 3.5724508050089447,
|
|
"step": 1000
|
|
},
|
|
{
|
|
"train_runtime": 2976.681,
|
|
"train_samples_per_second": 96.118,
|
|
"train_steps_per_second": 0.75,
|
|
"total_flos": 4.722006162277663e+17,
|
|
"train_loss": 0.444128730237484,
|
|
"epoch": 3.5724508050089447,
|
|
"step": 1000
|
|
}
|
|
],
|
|
"best_metric": 0.3169912099838257,
|
|
"best_model_checkpoint": "saves/adpr-llama/checkpoint-700",
|
|
"is_local_process_zero": true,
|
|
"is_world_process_zero": true,
|
|
"is_hyper_param_search": false,
|
|
"trial_name": null,
|
|
"trial_params": null,
|
|
"stateful_callbacks": {
|
|
"EarlyStoppingCallback": {
|
|
"args": {
|
|
"early_stopping_patience": 3,
|
|
"early_stopping_threshold": 0.0
|
|
},
|
|
"attributes": {
|
|
"early_stopping_patience_counter": 3
|
|
}
|
|
},
|
|
"TrainerControl": {
|
|
"args": {
|
|
"should_training_stop": true,
|
|
"should_epoch_stop": false,
|
|
"should_save": true,
|
|
"should_evaluate": false,
|
|
"should_log": false
|
|
},
|
|
"attributes": {}
|
|
}
|
|
},
|
|
"model": "PeftModelForCausalLM(\n (base_model): LoraModel(\n (model): LlamaForCausalLM(\n (model): LlamaModel(\n (embed_tokens): Embedding(32000, 4096, padding_idx=0)\n (layers): ModuleList(\n (0-31): 32 x LlamaDecoderLayer(\n (self_attn): LlamaAttention(\n (q_proj): lora.Linear(\n (base_layer): Linear(in_features=4096, out_features=4096, bias=False)\n (lora_dropout): ModuleDict(\n (default): Dropout(p=0.05, inplace=False)\n )\n (lora_A): ModuleDict(\n (default): Linear(in_features=4096, out_features=64, bias=False)\n )\n (lora_B): ModuleDict(\n (default): Linear(in_features=64, out_features=4096, bias=False)\n )\n (lora_embedding_A): ParameterDict()\n (lora_embedding_B): ParameterDict()\n (lora_magnitude_vector): ModuleDict()\n )\n (k_proj): lora.Linear(\n (base_layer): Linear(in_features=4096, out_features=4096, bias=False)\n (lora_dropout): ModuleDict(\n (default): Dropout(p=0.05, inplace=False)\n )\n (lora_A): ModuleDict(\n (default): Linear(in_features=4096, out_features=64, bias=False)\n )\n (lora_B): ModuleDict(\n (default): Linear(in_features=64, out_features=4096, bias=False)\n )\n (lora_embedding_A): ParameterDict()\n (lora_embedding_B): ParameterDict()\n (lora_magnitude_vector): ModuleDict()\n )\n (v_proj): lora.Linear(\n (base_layer): Linear(in_features=4096, out_features=4096, bias=False)\n (lora_dropout): ModuleDict(\n (default): Dropout(p=0.05, inplace=False)\n )\n (lora_A): ModuleDict(\n (default): Linear(in_features=4096, out_features=64, bias=False)\n )\n (lora_B): ModuleDict(\n (default): Linear(in_features=64, out_features=4096, bias=False)\n )\n (lora_embedding_A): ParameterDict()\n (lora_embedding_B): ParameterDict()\n (lora_magnitude_vector): ModuleDict()\n )\n (o_proj): lora.Linear(\n (base_layer): Linear(in_features=4096, out_features=4096, bias=False)\n (lora_dropout): ModuleDict(\n (default): Dropout(p=0.05, inplace=False)\n )\n (lora_A): ModuleDict(\n (default): Linear(in_features=4096, out_features=64, bias=False)\n )\n (lora_B): ModuleDict(\n (default): Linear(in_features=64, out_features=4096, bias=False)\n )\n (lora_embedding_A): ParameterDict()\n (lora_embedding_B): ParameterDict()\n (lora_magnitude_vector): ModuleDict()\n )\n )\n (mlp): LlamaMLP(\n (gate_proj): lora.Linear(\n (base_layer): Linear(in_features=4096, out_features=11008, bias=False)\n (lora_dropout): ModuleDict(\n (default): Dropout(p=0.05, inplace=False)\n )\n (lora_A): ModuleDict(\n (default): Linear(in_features=4096, out_features=64, bias=False)\n )\n (lora_B): ModuleDict(\n (default): Linear(in_features=64, out_features=11008, bias=False)\n )\n (lora_embedding_A): ParameterDict()\n (lora_embedding_B): ParameterDict()\n (lora_magnitude_vector): ModuleDict()\n )\n (up_proj): lora.Linear(\n (base_layer): Linear(in_features=4096, out_features=11008, bias=False)\n (lora_dropout): ModuleDict(\n (default): Dropout(p=0.05, inplace=False)\n )\n (lora_A): ModuleDict(\n (default): Linear(in_features=4096, out_features=64, bias=False)\n )\n (lora_B): ModuleDict(\n (default): Linear(in_features=64, out_features=11008, bias=False)\n )\n (lora_embedding_A): ParameterDict()\n (lora_embedding_B): ParameterDict()\n (lora_magnitude_vector): ModuleDict()\n )\n (down_proj): lora.Linear(\n (base_layer): Linear(in_features=11008, out_features=4096, bias=False)\n (lora_dropout): ModuleDict(\n (default): Dropout(p=0.05, inplace=False)\n )\n (lora_A): ModuleDict(\n (default): Linear(in_features=11008, out_features=64, bias=False)\n )\n (lora_B): ModuleDict(\n (default): Linear(in_features=64, out_features=4096, bias=False)\n )\n (lora_embedding_A): ParameterDict()\n (lora_embedding_B): ParameterDict()\n (lora_magnitude_vector): ModuleDict()\n )\n (act_fn): SiLU()\n )\n (input_layernorm): LlamaRMSNorm((4096,), eps=1e-05)\n (post_attention_layernorm): LlamaRMSNorm((4096,), eps=1e-05)\n )\n )\n (norm): LlamaRMSNorm((4096,), eps=1e-05)\n (rotary_emb): LlamaRotaryEmbedding()\n )\n (lm_head): Linear(in_features=4096, out_features=32000, bias=False)\n )\n )\n)",
|
|
"optimizer": "AcceleratedOptimizer (\nParameter Group 0\n amsgrad: False\n betas: (0.9, 0.999)\n capturable: False\n decoupled_weight_decay: True\n differentiable: False\n eps: 1e-08\n foreach: None\n fused: None\n initial_lr: 0.0003\n lr: 0.00017905266787499947\n maximize: False\n weight_decay: 0.0\n\nParameter Group 1\n amsgrad: False\n betas: (0.9, 0.999)\n capturable: False\n decoupled_weight_decay: True\n differentiable: False\n eps: 1e-08\n foreach: None\n fused: None\n initial_lr: 0.0003\n lr: 0.00017905266787499947\n maximize: False\n weight_decay: 0.0\n)",
|
|
"lr_scheduler": "<torch.optim.lr_scheduler.LambdaLR object at 0x7c03193e23f0>",
|
|
"train_dataloader": "<accelerate.data_loader.DataLoaderShard object at 0x7c03195b34d0>"
|
|
} |