280 lines
6.6 KiB
JSON
280 lines
6.6 KiB
JSON
{
|
|
"best_global_step": null,
|
|
"best_metric": null,
|
|
"best_model_checkpoint": null,
|
|
"epoch": 1.8005148005148004,
|
|
"eval_steps": 500,
|
|
"global_step": 700,
|
|
"is_hyper_param_search": false,
|
|
"is_local_process_zero": true,
|
|
"is_world_process_zero": true,
|
|
"log_history": [
|
|
{
|
|
"epoch": 0.05148005148005148,
|
|
"grad_norm": 3.7984347343444824,
|
|
"learning_rate": 2.435897435897436e-06,
|
|
"loss": 3.0721,
|
|
"step": 20
|
|
},
|
|
{
|
|
"epoch": 0.10296010296010295,
|
|
"grad_norm": 2.0532360076904297,
|
|
"learning_rate": 5e-06,
|
|
"loss": 2.7105,
|
|
"step": 40
|
|
},
|
|
{
|
|
"epoch": 0.15444015444015444,
|
|
"grad_norm": 1.889143466949463,
|
|
"learning_rate": 7.564102564102564e-06,
|
|
"loss": 2.397,
|
|
"step": 60
|
|
},
|
|
{
|
|
"epoch": 0.2059202059202059,
|
|
"grad_norm": 2.206676959991455,
|
|
"learning_rate": 9.999949644960027e-06,
|
|
"loss": 2.1587,
|
|
"step": 80
|
|
},
|
|
{
|
|
"epoch": 0.2574002574002574,
|
|
"grad_norm": 2.4184727668762207,
|
|
"learning_rate": 9.9778098230154e-06,
|
|
"loss": 1.9613,
|
|
"step": 100
|
|
},
|
|
{
|
|
"epoch": 0.3088803088803089,
|
|
"grad_norm": 2.1078503131866455,
|
|
"learning_rate": 9.915591603280632e-06,
|
|
"loss": 1.9187,
|
|
"step": 120
|
|
},
|
|
{
|
|
"epoch": 0.36036036036036034,
|
|
"grad_norm": 2.228259801864624,
|
|
"learning_rate": 9.813795930277306e-06,
|
|
"loss": 1.863,
|
|
"step": 140
|
|
},
|
|
{
|
|
"epoch": 0.4118404118404118,
|
|
"grad_norm": 3.990351438522339,
|
|
"learning_rate": 9.673242402909555e-06,
|
|
"loss": 1.7983,
|
|
"step": 160
|
|
},
|
|
{
|
|
"epoch": 0.46332046332046334,
|
|
"grad_norm": 2.6102797985076904,
|
|
"learning_rate": 9.495062675535614e-06,
|
|
"loss": 1.7845,
|
|
"step": 180
|
|
},
|
|
{
|
|
"epoch": 0.5148005148005148,
|
|
"grad_norm": 2.4415857791900635,
|
|
"learning_rate": 9.280691346552308e-06,
|
|
"loss": 1.7541,
|
|
"step": 200
|
|
},
|
|
{
|
|
"epoch": 0.5662805662805663,
|
|
"grad_norm": 2.875765800476074,
|
|
"learning_rate": 9.031854407852317e-06,
|
|
"loss": 1.7665,
|
|
"step": 220
|
|
},
|
|
{
|
|
"epoch": 0.6177606177606177,
|
|
"grad_norm": 2.55922269821167,
|
|
"learning_rate": 8.750555348152299e-06,
|
|
"loss": 1.7272,
|
|
"step": 240
|
|
},
|
|
{
|
|
"epoch": 0.6692406692406693,
|
|
"grad_norm": 3.4958205223083496,
|
|
"learning_rate": 8.439059022079789e-06,
|
|
"loss": 1.7349,
|
|
"step": 260
|
|
},
|
|
{
|
|
"epoch": 0.7207207207207207,
|
|
"grad_norm": 2.582611560821533,
|
|
"learning_rate": 8.099873414895453e-06,
|
|
"loss": 1.7273,
|
|
"step": 280
|
|
},
|
|
{
|
|
"epoch": 0.7722007722007722,
|
|
"grad_norm": 2.274104595184326,
|
|
"learning_rate": 7.73572944967043e-06,
|
|
"loss": 1.695,
|
|
"step": 300
|
|
},
|
|
{
|
|
"epoch": 0.8236808236808236,
|
|
"grad_norm": 2.8610410690307617,
|
|
"learning_rate": 7.3495589994995274e-06,
|
|
"loss": 1.663,
|
|
"step": 320
|
|
},
|
|
{
|
|
"epoch": 0.8751608751608752,
|
|
"grad_norm": 2.4936399459838867,
|
|
"learning_rate": 6.944471281782975e-06,
|
|
"loss": 1.6462,
|
|
"step": 340
|
|
},
|
|
{
|
|
"epoch": 0.9266409266409267,
|
|
"grad_norm": 3.510580062866211,
|
|
"learning_rate": 6.523727824636103e-06,
|
|
"loss": 1.6662,
|
|
"step": 360
|
|
},
|
|
{
|
|
"epoch": 0.9781209781209781,
|
|
"grad_norm": 2.398237943649292,
|
|
"learning_rate": 6.090716206982714e-06,
|
|
"loss": 1.6664,
|
|
"step": 380
|
|
},
|
|
{
|
|
"epoch": 1.0283140283140284,
|
|
"grad_norm": 2.8953514099121094,
|
|
"learning_rate": 5.648922783761443e-06,
|
|
"loss": 1.6363,
|
|
"step": 400
|
|
},
|
|
{
|
|
"epoch": 1.0797940797940797,
|
|
"grad_norm": 2.4946253299713135,
|
|
"learning_rate": 5.201904615845743e-06,
|
|
"loss": 1.5998,
|
|
"step": 420
|
|
},
|
|
{
|
|
"epoch": 1.1312741312741312,
|
|
"grad_norm": 2.965862512588501,
|
|
"learning_rate": 4.753260830681247e-06,
|
|
"loss": 1.5663,
|
|
"step": 440
|
|
},
|
|
{
|
|
"epoch": 1.1827541827541828,
|
|
"grad_norm": 2.843411922454834,
|
|
"learning_rate": 4.306603644227821e-06,
|
|
"loss": 1.578,
|
|
"step": 460
|
|
},
|
|
{
|
|
"epoch": 1.2342342342342343,
|
|
"grad_norm": 2.5591073036193848,
|
|
"learning_rate": 3.8655292775206185e-06,
|
|
"loss": 1.5821,
|
|
"step": 480
|
|
},
|
|
{
|
|
"epoch": 1.2857142857142856,
|
|
"grad_norm": 3.3472440242767334,
|
|
"learning_rate": 3.4335890020128382e-06,
|
|
"loss": 1.5743,
|
|
"step": 500
|
|
},
|
|
{
|
|
"epoch": 1.3371943371943371,
|
|
"grad_norm": 2.687406063079834,
|
|
"learning_rate": 3.0142605468260976e-06,
|
|
"loss": 1.5407,
|
|
"step": 520
|
|
},
|
|
{
|
|
"epoch": 1.3886743886743886,
|
|
"grad_norm": 2.7775464057922363,
|
|
"learning_rate": 2.610920098120424e-06,
|
|
"loss": 1.5429,
|
|
"step": 540
|
|
},
|
|
{
|
|
"epoch": 1.4401544401544402,
|
|
"grad_norm": 2.49112606048584,
|
|
"learning_rate": 2.2268151160284508e-06,
|
|
"loss": 1.5408,
|
|
"step": 560
|
|
},
|
|
{
|
|
"epoch": 1.4916344916344917,
|
|
"grad_norm": 2.938004493713379,
|
|
"learning_rate": 1.8650381880159108e-06,
|
|
"loss": 1.5605,
|
|
"step": 580
|
|
},
|
|
{
|
|
"epoch": 1.5431145431145432,
|
|
"grad_norm": 2.910888910293579,
|
|
"learning_rate": 1.5285021291857705e-06,
|
|
"loss": 1.53,
|
|
"step": 600
|
|
},
|
|
{
|
|
"epoch": 1.5945945945945947,
|
|
"grad_norm": 2.5734620094299316,
|
|
"learning_rate": 1.2199165300037358e-06,
|
|
"loss": 1.5425,
|
|
"step": 620
|
|
},
|
|
{
|
|
"epoch": 1.646074646074646,
|
|
"grad_norm": 2.9094061851501465,
|
|
"learning_rate": 9.417659402690254e-07,
|
|
"loss": 1.5401,
|
|
"step": 640
|
|
},
|
|
{
|
|
"epoch": 1.6975546975546976,
|
|
"grad_norm": 2.8157808780670166,
|
|
"learning_rate": 6.962898649802824e-07,
|
|
"loss": 1.5725,
|
|
"step": 660
|
|
},
|
|
{
|
|
"epoch": 1.7490347490347489,
|
|
"grad_norm": 4.436152458190918,
|
|
"learning_rate": 4.854647331581385e-07,
|
|
"loss": 1.5144,
|
|
"step": 680
|
|
},
|
|
{
|
|
"epoch": 1.8005148005148004,
|
|
"grad_norm": 2.858851909637451,
|
|
"learning_rate": 3.109879848008557e-07,
|
|
"loss": 1.5597,
|
|
"step": 700
|
|
}
|
|
],
|
|
"logging_steps": 20,
|
|
"max_steps": 778,
|
|
"num_input_tokens_seen": 0,
|
|
"num_train_epochs": 2,
|
|
"save_steps": 50,
|
|
"stateful_callbacks": {
|
|
"TrainerControl": {
|
|
"args": {
|
|
"should_epoch_stop": false,
|
|
"should_evaluate": false,
|
|
"should_log": false,
|
|
"should_save": true,
|
|
"should_training_stop": false
|
|
},
|
|
"attributes": {}
|
|
}
|
|
},
|
|
"total_flos": 5.112035123977421e+16,
|
|
"train_batch_size": 4,
|
|
"trial_name": null,
|
|
"trial_params": null
|
|
}
|