976 lines
31 KiB
JSON
976 lines
31 KiB
JSON
{
|
|
"best_global_step": null,
|
|
"best_metric": null,
|
|
"best_model_checkpoint": null,
|
|
"epoch": 2.526582278481013,
|
|
"eval_steps": 50,
|
|
"global_step": 500,
|
|
"is_hyper_param_search": false,
|
|
"is_local_process_zero": true,
|
|
"is_world_process_zero": true,
|
|
"log_history": [
|
|
{
|
|
"epoch": 0,
|
|
"eval_indic_sft_mini_val_loss": 1.127184510231018,
|
|
"eval_indic_sft_mini_val_runtime": 156.0033,
|
|
"eval_indic_sft_mini_val_samples_per_second": 12.307,
|
|
"eval_indic_sft_mini_val_steps_per_second": 3.077,
|
|
"memory/device_reserved (GiB)": 24.22,
|
|
"memory/max_active (GiB)": 24.14,
|
|
"memory/max_allocated (GiB)": 24.14,
|
|
"step": 0
|
|
},
|
|
{
|
|
"epoch": 0,
|
|
"eval_tulu_sft_mini_val_loss": 2.330348491668701,
|
|
"eval_tulu_sft_mini_val_runtime": 91.8069,
|
|
"eval_tulu_sft_mini_val_samples_per_second": 12.548,
|
|
"eval_tulu_sft_mini_val_steps_per_second": 3.137,
|
|
"memory/device_reserved (GiB)": 24.22,
|
|
"memory/max_active (GiB)": 24.14,
|
|
"memory/max_allocated (GiB)": 24.14,
|
|
"step": 0
|
|
},
|
|
{
|
|
"epoch": 0.05063291139240506,
|
|
"grad_norm": 1.9375,
|
|
"learning_rate": 1.0588235294117648e-05,
|
|
"loss": 1.2239381790161132,
|
|
"memory/device_reserved (GiB)": 37.02,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 3.40055,
|
|
"step": 10,
|
|
"tokens/total": 1310720,
|
|
"tokens/trainable": 1268231
|
|
},
|
|
{
|
|
"epoch": 0.10126582278481013,
|
|
"grad_norm": 1.640625,
|
|
"learning_rate": 1.9999400896826965e-05,
|
|
"loss": 1.1529170989990234,
|
|
"memory/device_reserved (GiB)": 37.02,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 3.16742,
|
|
"step": 20,
|
|
"tokens/total": 2621440,
|
|
"tokens/train_per_sec_per_gpu": 17988.68,
|
|
"tokens/trainable": 2535045
|
|
},
|
|
{
|
|
"epoch": 0.1518987341772152,
|
|
"grad_norm": 1.4453125,
|
|
"learning_rate": 1.9978439822224228e-05,
|
|
"loss": 1.1212153434753418,
|
|
"memory/device_reserved (GiB)": 37.02,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 3.06858,
|
|
"step": 30,
|
|
"tokens/total": 3932160,
|
|
"tokens/train_per_sec_per_gpu": 17958.15,
|
|
"tokens/trainable": 3804033
|
|
},
|
|
{
|
|
"epoch": 0.20253164556962025,
|
|
"grad_norm": 1.3828125,
|
|
"learning_rate": 1.9927595335238736e-05,
|
|
"loss": 1.106326198577881,
|
|
"memory/device_reserved (GiB)": 37.02,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 3.02323,
|
|
"step": 40,
|
|
"tokens/total": 5242880,
|
|
"tokens/train_per_sec_per_gpu": 17929.18,
|
|
"tokens/trainable": 5070883
|
|
},
|
|
{
|
|
"epoch": 0.25316455696202533,
|
|
"grad_norm": 1.421875,
|
|
"learning_rate": 1.984701970484229e-05,
|
|
"loss": 1.0841044425964355,
|
|
"memory/device_reserved (GiB)": 37.02,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.95679,
|
|
"step": 50,
|
|
"tokens/total": 6553600,
|
|
"tokens/train_per_sec_per_gpu": 17943.82,
|
|
"tokens/trainable": 6341109
|
|
},
|
|
{
|
|
"epoch": 0.25316455696202533,
|
|
"eval_indic_sft_mini_val_loss": 1.0926785469055176,
|
|
"eval_indic_sft_mini_val_runtime": 157.8472,
|
|
"eval_indic_sft_mini_val_samples_per_second": 12.164,
|
|
"eval_indic_sft_mini_val_steps_per_second": 3.041,
|
|
"memory/device_reserved (GiB)": 37.02,
|
|
"memory/max_active (GiB)": 25.99,
|
|
"memory/max_allocated (GiB)": 25.99,
|
|
"step": 50
|
|
},
|
|
{
|
|
"epoch": 0.25316455696202533,
|
|
"eval_tulu_sft_mini_val_loss": 2.3083696365356445,
|
|
"eval_tulu_sft_mini_val_runtime": 92.1987,
|
|
"eval_tulu_sft_mini_val_samples_per_second": 12.495,
|
|
"eval_tulu_sft_mini_val_steps_per_second": 3.124,
|
|
"memory/device_reserved (GiB)": 26.58,
|
|
"memory/max_active (GiB)": 25.99,
|
|
"memory/max_allocated (GiB)": 25.99,
|
|
"step": 50
|
|
},
|
|
{
|
|
"epoch": 0.3037974683544304,
|
|
"grad_norm": 1.34375,
|
|
"learning_rate": 1.9736954238777793e-05,
|
|
"loss": 1.1303999900817872,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 3.09689,
|
|
"step": 60,
|
|
"tokens/total": 7864320,
|
|
"tokens/train_per_sec_per_gpu": 3953.8,
|
|
"tokens/trainable": 7607732
|
|
},
|
|
{
|
|
"epoch": 0.35443037974683544,
|
|
"grad_norm": 1.4296875,
|
|
"learning_rate": 1.9597728560891266e-05,
|
|
"loss": 1.1227096557617187,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 3.07317,
|
|
"step": 70,
|
|
"tokens/total": 9175040,
|
|
"tokens/train_per_sec_per_gpu": 17981.63,
|
|
"tokens/trainable": 8877038
|
|
},
|
|
{
|
|
"epoch": 0.4050632911392405,
|
|
"grad_norm": 1.3828125,
|
|
"learning_rate": 1.9429759623974992e-05,
|
|
"loss": 1.109241771697998,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 3.03206,
|
|
"step": 80,
|
|
"tokens/total": 10485760,
|
|
"tokens/train_per_sec_per_gpu": 17934.18,
|
|
"tokens/trainable": 10145302
|
|
},
|
|
{
|
|
"epoch": 0.45569620253164556,
|
|
"grad_norm": 1.2578125,
|
|
"learning_rate": 1.9233550461078114e-05,
|
|
"loss": 1.071034049987793,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.9184,
|
|
"step": 90,
|
|
"tokens/total": 11796480,
|
|
"tokens/train_per_sec_per_gpu": 17952.48,
|
|
"tokens/trainable": 11415878
|
|
},
|
|
{
|
|
"epoch": 0.5063291139240507,
|
|
"grad_norm": 1.2421875,
|
|
"learning_rate": 1.900968867902419e-05,
|
|
"loss": 1.0816166877746582,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.94944,
|
|
"step": 100,
|
|
"tokens/total": 13107200,
|
|
"tokens/train_per_sec_per_gpu": 17943.02,
|
|
"tokens/trainable": 12684573
|
|
},
|
|
{
|
|
"epoch": 0.5063291139240507,
|
|
"eval_indic_sft_mini_val_loss": 1.0623797178268433,
|
|
"eval_indic_sft_mini_val_runtime": 157.7495,
|
|
"eval_indic_sft_mini_val_samples_per_second": 12.171,
|
|
"eval_indic_sft_mini_val_steps_per_second": 3.043,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 25.99,
|
|
"memory/max_allocated (GiB)": 25.99,
|
|
"step": 100
|
|
},
|
|
{
|
|
"epoch": 0.5063291139240507,
|
|
"eval_tulu_sft_mini_val_loss": 2.2967746257781982,
|
|
"eval_tulu_sft_mini_val_runtime": 92.342,
|
|
"eval_tulu_sft_mini_val_samples_per_second": 12.475,
|
|
"eval_tulu_sft_mini_val_steps_per_second": 3.119,
|
|
"memory/device_reserved (GiB)": 26.58,
|
|
"memory/max_active (GiB)": 25.99,
|
|
"memory/max_allocated (GiB)": 25.99,
|
|
"step": 100
|
|
},
|
|
{
|
|
"epoch": 0.5569620253164557,
|
|
"grad_norm": 1.2421875,
|
|
"learning_rate": 1.8758844698647457e-05,
|
|
"loss": 1.106839370727539,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 3.02478,
|
|
"step": 110,
|
|
"tokens/total": 14417920,
|
|
"tokens/train_per_sec_per_gpu": 3955.64,
|
|
"tokens/trainable": 13951995
|
|
},
|
|
{
|
|
"epoch": 0.6075949367088608,
|
|
"grad_norm": 1.8984375,
|
|
"learning_rate": 1.848176974701775e-05,
|
|
"loss": 1.0786801338195802,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.9408,
|
|
"step": 120,
|
|
"tokens/total": 15728640,
|
|
"tokens/train_per_sec_per_gpu": 17975.05,
|
|
"tokens/trainable": 15221447
|
|
},
|
|
{
|
|
"epoch": 0.6582278481012658,
|
|
"grad_norm": 1.28125,
|
|
"learning_rate": 1.8179293607667177e-05,
|
|
"loss": 1.0739567756652832,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.92694,
|
|
"step": 130,
|
|
"tokens/total": 17039360,
|
|
"tokens/train_per_sec_per_gpu": 17954.31,
|
|
"tokens/trainable": 16490903
|
|
},
|
|
{
|
|
"epoch": 0.7088607594936709,
|
|
"grad_norm": 1.2890625,
|
|
"learning_rate": 1.7852322135555946e-05,
|
|
"loss": 1.0759190559387206,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.93269,
|
|
"step": 140,
|
|
"tokens/total": 18350080,
|
|
"tokens/train_per_sec_per_gpu": 17887.36,
|
|
"tokens/trainable": 17757184
|
|
},
|
|
{
|
|
"epoch": 0.759493670886076,
|
|
"grad_norm": 1.3828125,
|
|
"learning_rate": 1.7501834544219697e-05,
|
|
"loss": 1.0366504669189454,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.81976,
|
|
"step": 150,
|
|
"tokens/total": 19660800,
|
|
"tokens/train_per_sec_per_gpu": 17911.77,
|
|
"tokens/trainable": 19024962
|
|
},
|
|
{
|
|
"epoch": 0.759493670886076,
|
|
"eval_indic_sft_mini_val_loss": 1.0453351736068726,
|
|
"eval_indic_sft_mini_val_runtime": 157.7623,
|
|
"eval_indic_sft_mini_val_samples_per_second": 12.17,
|
|
"eval_indic_sft_mini_val_steps_per_second": 3.043,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 25.99,
|
|
"memory/max_allocated (GiB)": 25.99,
|
|
"step": 150
|
|
},
|
|
{
|
|
"epoch": 0.759493670886076,
|
|
"eval_tulu_sft_mini_val_loss": 2.2883615493774414,
|
|
"eval_tulu_sft_mini_val_runtime": 92.4616,
|
|
"eval_tulu_sft_mini_val_samples_per_second": 12.459,
|
|
"eval_tulu_sft_mini_val_steps_per_second": 3.115,
|
|
"memory/device_reserved (GiB)": 26.58,
|
|
"memory/max_active (GiB)": 25.99,
|
|
"memory/max_allocated (GiB)": 25.99,
|
|
"step": 150
|
|
},
|
|
{
|
|
"epoch": 0.810126582278481,
|
|
"grad_norm": 1.328125,
|
|
"learning_rate": 1.7128880473222688e-05,
|
|
"loss": 1.0812637329101562,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.9484,
|
|
"step": 160,
|
|
"tokens/total": 20971520,
|
|
"tokens/train_per_sec_per_gpu": 3951.59,
|
|
"tokens/trainable": 20291324
|
|
},
|
|
{
|
|
"epoch": 0.8607594936708861,
|
|
"grad_norm": 1.328125,
|
|
"learning_rate": 1.6734576844699234e-05,
|
|
"loss": 1.0667606353759767,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.90595,
|
|
"step": 170,
|
|
"tokens/total": 22282240,
|
|
"tokens/train_per_sec_per_gpu": 17990.97,
|
|
"tokens/trainable": 21559984
|
|
},
|
|
{
|
|
"epoch": 0.9113924050632911,
|
|
"grad_norm": 1.3515625,
|
|
"learning_rate": 1.6320104518397473e-05,
|
|
"loss": 1.067934513092041,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.90936,
|
|
"step": 180,
|
|
"tokens/total": 23592960,
|
|
"tokens/train_per_sec_per_gpu": 17944.73,
|
|
"tokens/trainable": 22828372
|
|
},
|
|
{
|
|
"epoch": 0.9620253164556962,
|
|
"grad_norm": 1.296875,
|
|
"learning_rate": 1.588670475524283e-05,
|
|
"loss": 1.04850492477417,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.85338,
|
|
"step": 190,
|
|
"tokens/total": 24903680,
|
|
"tokens/train_per_sec_per_gpu": 17890.69,
|
|
"tokens/trainable": 24094636
|
|
},
|
|
{
|
|
"epoch": 1.010126582278481,
|
|
"grad_norm": 1.3046875,
|
|
"learning_rate": 1.5435675500012212e-05,
|
|
"loss": 1.0491924285888672,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.85534,
|
|
"step": 200,
|
|
"tokens/total": 26136576,
|
|
"tokens/train_per_sec_per_gpu": 17465.18,
|
|
"tokens/trainable": 25286404
|
|
},
|
|
{
|
|
"epoch": 1.010126582278481,
|
|
"eval_indic_sft_mini_val_loss": 1.0354256629943848,
|
|
"eval_indic_sft_mini_val_runtime": 157.741,
|
|
"eval_indic_sft_mini_val_samples_per_second": 12.172,
|
|
"eval_indic_sft_mini_val_steps_per_second": 3.043,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 25.99,
|
|
"memory/max_allocated (GiB)": 25.99,
|
|
"step": 200
|
|
},
|
|
{
|
|
"epoch": 1.010126582278481,
|
|
"eval_tulu_sft_mini_val_loss": 2.291062593460083,
|
|
"eval_tulu_sft_mini_val_runtime": 92.344,
|
|
"eval_tulu_sft_mini_val_samples_per_second": 12.475,
|
|
"eval_tulu_sft_mini_val_steps_per_second": 3.119,
|
|
"memory/device_reserved (GiB)": 26.58,
|
|
"memory/max_active (GiB)": 25.99,
|
|
"memory/max_allocated (GiB)": 25.99,
|
|
"step": 200
|
|
},
|
|
{
|
|
"epoch": 1.0607594936708862,
|
|
"grad_norm": 1.375,
|
|
"learning_rate": 1.4968367494251486e-05,
|
|
"loss": 1.0487144470214844,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.85398,
|
|
"step": 210,
|
|
"tokens/total": 27447296,
|
|
"tokens/train_per_sec_per_gpu": 3957.26,
|
|
"tokens/trainable": 26554084
|
|
},
|
|
{
|
|
"epoch": 1.111392405063291,
|
|
"grad_norm": 1.3671875,
|
|
"learning_rate": 1.4486180231077278e-05,
|
|
"loss": 1.0355692863464356,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.81671,
|
|
"step": 220,
|
|
"tokens/total": 28758016,
|
|
"tokens/train_per_sec_per_gpu": 17979.8,
|
|
"tokens/trainable": 27823280
|
|
},
|
|
{
|
|
"epoch": 1.1620253164556962,
|
|
"grad_norm": 1.265625,
|
|
"learning_rate": 1.3990557763977694e-05,
|
|
"loss": 1.0380287170410156,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.82365,
|
|
"step": 230,
|
|
"tokens/total": 30068736,
|
|
"tokens/train_per_sec_per_gpu": 17964.75,
|
|
"tokens/trainable": 29092706
|
|
},
|
|
{
|
|
"epoch": 1.2126582278481013,
|
|
"grad_norm": 1.2734375,
|
|
"learning_rate": 1.3482984382163713e-05,
|
|
"loss": 1.036821174621582,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.82024,
|
|
"step": 240,
|
|
"tokens/total": 31379456,
|
|
"tokens/train_per_sec_per_gpu": 17930.26,
|
|
"tokens/trainable": 30360732
|
|
},
|
|
{
|
|
"epoch": 1.2632911392405064,
|
|
"grad_norm": 1.2265625,
|
|
"learning_rate": 1.2964980165422701e-05,
|
|
"loss": 0.995778751373291,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.70683,
|
|
"step": 250,
|
|
"tokens/total": 32690176,
|
|
"tokens/train_per_sec_per_gpu": 17927.94,
|
|
"tokens/trainable": 31631352
|
|
},
|
|
{
|
|
"epoch": 1.2632911392405064,
|
|
"eval_indic_sft_mini_val_loss": 1.027944803237915,
|
|
"eval_indic_sft_mini_val_runtime": 157.8775,
|
|
"eval_indic_sft_mini_val_samples_per_second": 12.161,
|
|
"eval_indic_sft_mini_val_steps_per_second": 3.04,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 25.99,
|
|
"memory/max_allocated (GiB)": 25.99,
|
|
"step": 250
|
|
},
|
|
{
|
|
"epoch": 1.2632911392405064,
|
|
"eval_tulu_sft_mini_val_loss": 2.2984113693237305,
|
|
"eval_tulu_sft_mini_val_runtime": 92.3048,
|
|
"eval_tulu_sft_mini_val_samples_per_second": 12.48,
|
|
"eval_tulu_sft_mini_val_steps_per_second": 3.12,
|
|
"memory/device_reserved (GiB)": 26.58,
|
|
"memory/max_active (GiB)": 25.99,
|
|
"memory/max_allocated (GiB)": 25.99,
|
|
"step": 250
|
|
},
|
|
{
|
|
"epoch": 1.3139240506329113,
|
|
"grad_norm": 1.2109375,
|
|
"learning_rate": 1.2438096431786408e-05,
|
|
"loss": 1.0438777923583984,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.84021,
|
|
"step": 260,
|
|
"tokens/total": 34000896,
|
|
"tokens/train_per_sec_per_gpu": 3955.89,
|
|
"tokens/trainable": 32899468
|
|
},
|
|
{
|
|
"epoch": 1.3645569620253164,
|
|
"grad_norm": 1.28125,
|
|
"learning_rate": 1.1903911091646684e-05,
|
|
"loss": 1.0240283012390137,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.78439,
|
|
"step": 270,
|
|
"tokens/total": 35311616,
|
|
"tokens/train_per_sec_per_gpu": 17961.08,
|
|
"tokens/trainable": 34168384
|
|
},
|
|
{
|
|
"epoch": 1.4151898734177215,
|
|
"grad_norm": 1.234375,
|
|
"learning_rate": 1.1364023922232503e-05,
|
|
"loss": 1.031261920928955,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.8046,
|
|
"step": 280,
|
|
"tokens/total": 36622336,
|
|
"tokens/train_per_sec_per_gpu": 17939.56,
|
|
"tokens/trainable": 35436704
|
|
},
|
|
{
|
|
"epoch": 1.4658227848101266,
|
|
"grad_norm": 1.3359375,
|
|
"learning_rate": 1.0820051776600175e-05,
|
|
"loss": 1.0369884490966796,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.82071,
|
|
"step": 290,
|
|
"tokens/total": 37933056,
|
|
"tokens/train_per_sec_per_gpu": 17936.98,
|
|
"tokens/trainable": 36705368
|
|
},
|
|
{
|
|
"epoch": 1.5164556962025317,
|
|
"grad_norm": 1.1875,
|
|
"learning_rate": 1.0273623741484924e-05,
|
|
"loss": 1.0238310813903808,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.78384,
|
|
"step": 300,
|
|
"tokens/total": 39243776,
|
|
"tokens/train_per_sec_per_gpu": 17926.64,
|
|
"tokens/trainable": 37973400
|
|
},
|
|
{
|
|
"epoch": 1.5164556962025317,
|
|
"eval_indic_sft_mini_val_loss": 1.0224663019180298,
|
|
"eval_indic_sft_mini_val_runtime": 157.771,
|
|
"eval_indic_sft_mini_val_samples_per_second": 12.17,
|
|
"eval_indic_sft_mini_val_steps_per_second": 3.042,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 25.99,
|
|
"memory/max_allocated (GiB)": 25.99,
|
|
"step": 300
|
|
},
|
|
{
|
|
"epoch": 1.5164556962025317,
|
|
"eval_tulu_sft_mini_val_loss": 2.295070171356201,
|
|
"eval_tulu_sft_mini_val_runtime": 92.349,
|
|
"eval_tulu_sft_mini_val_samples_per_second": 12.474,
|
|
"eval_tulu_sft_mini_val_steps_per_second": 3.119,
|
|
"memory/device_reserved (GiB)": 26.58,
|
|
"memory/max_active (GiB)": 25.99,
|
|
"memory/max_allocated (GiB)": 25.99,
|
|
"step": 300
|
|
},
|
|
{
|
|
"epoch": 1.5670886075949366,
|
|
"grad_norm": 1.2421875,
|
|
"learning_rate": 9.726376258515077e-06,
|
|
"loss": 1.0644823074340821,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.89934,
|
|
"step": 310,
|
|
"tokens/total": 40554496,
|
|
"tokens/train_per_sec_per_gpu": 3956.31,
|
|
"tokens/trainable": 39240980
|
|
},
|
|
{
|
|
"epoch": 1.6177215189873417,
|
|
"grad_norm": 1.234375,
|
|
"learning_rate": 9.179948223399828e-06,
|
|
"loss": 1.0305088996887206,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.80249,
|
|
"step": 320,
|
|
"tokens/total": 41865216,
|
|
"tokens/train_per_sec_per_gpu": 17965.06,
|
|
"tokens/trainable": 40509632
|
|
},
|
|
{
|
|
"epoch": 1.6683544303797468,
|
|
"grad_norm": 1.265625,
|
|
"learning_rate": 8.6359760777675e-06,
|
|
"loss": 1.0250298500061035,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.78718,
|
|
"step": 330,
|
|
"tokens/total": 43175936,
|
|
"tokens/train_per_sec_per_gpu": 17953.88,
|
|
"tokens/trainable": 41777768
|
|
},
|
|
{
|
|
"epoch": 1.7189873417721517,
|
|
"grad_norm": 1.2265625,
|
|
"learning_rate": 8.096088908353316e-06,
|
|
"loss": 1.0106207847595214,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.74731,
|
|
"step": 340,
|
|
"tokens/total": 44486656,
|
|
"tokens/train_per_sec_per_gpu": 17933.81,
|
|
"tokens/trainable": 43045456
|
|
},
|
|
{
|
|
"epoch": 1.769620253164557,
|
|
"grad_norm": 1.2578125,
|
|
"learning_rate": 7.561903568213595e-06,
|
|
"loss": 0.9956131935119629,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.70638,
|
|
"step": 350,
|
|
"tokens/total": 45797376,
|
|
"tokens/train_per_sec_per_gpu": 17927.21,
|
|
"tokens/trainable": 44311528
|
|
},
|
|
{
|
|
"epoch": 1.769620253164557,
|
|
"eval_indic_sft_mini_val_loss": 1.0202709436416626,
|
|
"eval_indic_sft_mini_val_runtime": 157.7597,
|
|
"eval_indic_sft_mini_val_samples_per_second": 12.17,
|
|
"eval_indic_sft_mini_val_steps_per_second": 3.043,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 25.99,
|
|
"memory/max_allocated (GiB)": 25.99,
|
|
"step": 350
|
|
},
|
|
{
|
|
"epoch": 1.769620253164557,
|
|
"eval_tulu_sft_mini_val_loss": 2.2989728450775146,
|
|
"eval_tulu_sft_mini_val_runtime": 92.4784,
|
|
"eval_tulu_sft_mini_val_samples_per_second": 12.457,
|
|
"eval_tulu_sft_mini_val_steps_per_second": 3.114,
|
|
"memory/device_reserved (GiB)": 26.58,
|
|
"memory/max_active (GiB)": 25.99,
|
|
"memory/max_allocated (GiB)": 25.99,
|
|
"step": 350
|
|
},
|
|
{
|
|
"epoch": 1.820253164556962,
|
|
"grad_norm": 1.2734375,
|
|
"learning_rate": 7.035019834577301e-06,
|
|
"loss": 1.0437658309936524,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.83989,
|
|
"step": 360,
|
|
"tokens/total": 47108096,
|
|
"tokens/train_per_sec_per_gpu": 3952.58,
|
|
"tokens/trainable": 45578288
|
|
},
|
|
{
|
|
"epoch": 1.870886075949367,
|
|
"grad_norm": 1.2265625,
|
|
"learning_rate": 6.517015617836292e-06,
|
|
"loss": 1.0040513038635255,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.72932,
|
|
"step": 370,
|
|
"tokens/total": 48418816,
|
|
"tokens/train_per_sec_per_gpu": 17939.3,
|
|
"tokens/trainable": 46846324
|
|
},
|
|
{
|
|
"epoch": 1.9215189873417722,
|
|
"grad_norm": 1.234375,
|
|
"learning_rate": 6.009442236022307e-06,
|
|
"loss": 1.0151902198791505,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.75989,
|
|
"step": 380,
|
|
"tokens/total": 49729536,
|
|
"tokens/train_per_sec_per_gpu": 17915.04,
|
|
"tokens/trainable": 48113428
|
|
},
|
|
{
|
|
"epoch": 1.972151898734177,
|
|
"grad_norm": 1.3125,
|
|
"learning_rate": 5.513819768922723e-06,
|
|
"loss": 0.997506046295166,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.71151,
|
|
"step": 390,
|
|
"tokens/total": 51040256,
|
|
"tokens/train_per_sec_per_gpu": 17963.73,
|
|
"tokens/trainable": 49382776
|
|
},
|
|
{
|
|
"epoch": 2.020253164556962,
|
|
"grad_norm": 1.2421875,
|
|
"learning_rate": 5.031632505748516e-06,
|
|
"loss": 1.0011167526245117,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.72132,
|
|
"step": 400,
|
|
"tokens/total": 52273152,
|
|
"tokens/train_per_sec_per_gpu": 17458.66,
|
|
"tokens/trainable": 50572704
|
|
},
|
|
{
|
|
"epoch": 2.020253164556962,
|
|
"eval_indic_sft_mini_val_loss": 1.0187087059020996,
|
|
"eval_indic_sft_mini_val_runtime": 158.0242,
|
|
"eval_indic_sft_mini_val_samples_per_second": 12.15,
|
|
"eval_indic_sft_mini_val_steps_per_second": 3.038,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 25.99,
|
|
"memory/max_allocated (GiB)": 25.99,
|
|
"step": 400
|
|
},
|
|
{
|
|
"epoch": 2.020253164556962,
|
|
"eval_tulu_sft_mini_val_loss": 2.2966930866241455,
|
|
"eval_tulu_sft_mini_val_runtime": 92.3222,
|
|
"eval_tulu_sft_mini_val_samples_per_second": 12.478,
|
|
"eval_tulu_sft_mini_val_steps_per_second": 3.12,
|
|
"memory/device_reserved (GiB)": 26.58,
|
|
"memory/max_active (GiB)": 25.99,
|
|
"memory/max_allocated (GiB)": 25.99,
|
|
"step": 400
|
|
},
|
|
{
|
|
"epoch": 2.070886075949367,
|
|
"grad_norm": 1.2421875,
|
|
"learning_rate": 4.56432449998779e-06,
|
|
"loss": 1.0103323936462403,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.74651,
|
|
"step": 410,
|
|
"tokens/total": 53583872,
|
|
"tokens/train_per_sec_per_gpu": 3952.63,
|
|
"tokens/trainable": 51839796
|
|
},
|
|
{
|
|
"epoch": 2.1215189873417724,
|
|
"grad_norm": 1.234375,
|
|
"learning_rate": 4.113295244757171e-06,
|
|
"loss": 1.0133058547973632,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.75469,
|
|
"step": 420,
|
|
"tokens/total": 54894592,
|
|
"tokens/train_per_sec_per_gpu": 17984.79,
|
|
"tokens/trainable": 53108816
|
|
},
|
|
{
|
|
"epoch": 2.1721518987341772,
|
|
"grad_norm": 1.21875,
|
|
"learning_rate": 3.679895481602529e-06,
|
|
"loss": 1.0214984893798829,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.77735,
|
|
"step": 430,
|
|
"tokens/total": 56205312,
|
|
"tokens/train_per_sec_per_gpu": 17959.57,
|
|
"tokens/trainable": 54378168
|
|
},
|
|
{
|
|
"epoch": 2.222784810126582,
|
|
"grad_norm": 1.2109375,
|
|
"learning_rate": 3.2654231553007665e-06,
|
|
"loss": 0.9840593338012695,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.67529,
|
|
"step": 440,
|
|
"tokens/total": 57516032,
|
|
"tokens/train_per_sec_per_gpu": 17930.45,
|
|
"tokens/trainable": 55645416
|
|
},
|
|
{
|
|
"epoch": 2.2734177215189875,
|
|
"grad_norm": 1.1953125,
|
|
"learning_rate": 2.871119526777315e-06,
|
|
"loss": 1.028823184967041,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.79777,
|
|
"step": 450,
|
|
"tokens/total": 58826752,
|
|
"tokens/train_per_sec_per_gpu": 17944.17,
|
|
"tokens/trainable": 56914256
|
|
},
|
|
{
|
|
"epoch": 2.2734177215189875,
|
|
"eval_indic_sft_mini_val_loss": 1.0182074308395386,
|
|
"eval_indic_sft_mini_val_runtime": 157.9039,
|
|
"eval_indic_sft_mini_val_samples_per_second": 12.159,
|
|
"eval_indic_sft_mini_val_steps_per_second": 3.04,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 25.99,
|
|
"memory/max_allocated (GiB)": 25.99,
|
|
"step": 450
|
|
},
|
|
{
|
|
"epoch": 2.2734177215189875,
|
|
"eval_tulu_sft_mini_val_loss": 2.297419309616089,
|
|
"eval_tulu_sft_mini_val_runtime": 92.4484,
|
|
"eval_tulu_sft_mini_val_samples_per_second": 12.461,
|
|
"eval_tulu_sft_mini_val_steps_per_second": 3.115,
|
|
"memory/device_reserved (GiB)": 26.59,
|
|
"memory/max_active (GiB)": 25.99,
|
|
"memory/max_allocated (GiB)": 25.99,
|
|
"step": 450
|
|
},
|
|
{
|
|
"epoch": 2.3240506329113924,
|
|
"grad_norm": 1.21875,
|
|
"learning_rate": 2.4981654557803026e-06,
|
|
"loss": 1.0112363815307617,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.749,
|
|
"step": 460,
|
|
"tokens/total": 60137472,
|
|
"tokens/train_per_sec_per_gpu": 3954.28,
|
|
"tokens/trainable": 58182604
|
|
},
|
|
{
|
|
"epoch": 2.3746835443037977,
|
|
"grad_norm": 1.265625,
|
|
"learning_rate": 2.1476778644440553e-06,
|
|
"loss": 0.9997093200683593,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.71749,
|
|
"step": 470,
|
|
"tokens/total": 61448192,
|
|
"tokens/train_per_sec_per_gpu": 17956.34,
|
|
"tokens/trainable": 59450748
|
|
},
|
|
{
|
|
"epoch": 2.4253164556962026,
|
|
"grad_norm": 1.21875,
|
|
"learning_rate": 1.820706392332824e-06,
|
|
"loss": 1.0180435180664062,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.76777,
|
|
"step": 480,
|
|
"tokens/total": 62758912,
|
|
"tokens/train_per_sec_per_gpu": 17939.84,
|
|
"tokens/trainable": 60720528
|
|
},
|
|
{
|
|
"epoch": 2.4759493670886075,
|
|
"grad_norm": 1.234375,
|
|
"learning_rate": 1.518230252982248e-06,
|
|
"loss": 1.040645217895508,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.83104,
|
|
"step": 490,
|
|
"tokens/total": 64069632,
|
|
"tokens/train_per_sec_per_gpu": 17931.8,
|
|
"tokens/trainable": 61988728
|
|
},
|
|
{
|
|
"epoch": 2.526582278481013,
|
|
"grad_norm": 1.171875,
|
|
"learning_rate": 1.2411553013525457e-06,
|
|
"loss": 1.0239150047302246,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 32.29,
|
|
"memory/max_allocated (GiB)": 32.29,
|
|
"ppl": 2.78407,
|
|
"step": 500,
|
|
"tokens/total": 65380352,
|
|
"tokens/train_per_sec_per_gpu": 17916.35,
|
|
"tokens/trainable": 63254896
|
|
},
|
|
{
|
|
"epoch": 2.526582278481013,
|
|
"eval_indic_sft_mini_val_loss": 1.018338680267334,
|
|
"eval_indic_sft_mini_val_runtime": 157.6995,
|
|
"eval_indic_sft_mini_val_samples_per_second": 12.175,
|
|
"eval_indic_sft_mini_val_steps_per_second": 3.044,
|
|
"memory/device_reserved (GiB)": 37.04,
|
|
"memory/max_active (GiB)": 25.99,
|
|
"memory/max_allocated (GiB)": 25.99,
|
|
"step": 500
|
|
},
|
|
{
|
|
"epoch": 2.526582278481013,
|
|
"eval_tulu_sft_mini_val_loss": 2.299147605895996,
|
|
"eval_tulu_sft_mini_val_runtime": 92.1488,
|
|
"eval_tulu_sft_mini_val_samples_per_second": 12.502,
|
|
"eval_tulu_sft_mini_val_steps_per_second": 3.125,
|
|
"memory/device_reserved (GiB)": 26.59,
|
|
"memory/max_active (GiB)": 25.99,
|
|
"memory/max_allocated (GiB)": 25.99,
|
|
"step": 500
|
|
}
|
|
],
|
|
"logging_steps": 10,
|
|
"max_steps": 591,
|
|
"num_input_tokens_seen": 0,
|
|
"num_train_epochs": 3,
|
|
"save_steps": 500,
|
|
"stateful_callbacks": {
|
|
"TrainerControl": {
|
|
"args": {
|
|
"should_epoch_stop": false,
|
|
"should_evaluate": false,
|
|
"should_log": false,
|
|
"should_save": true,
|
|
"should_training_stop": false
|
|
},
|
|
"attributes": {}
|
|
}
|
|
},
|
|
"total_flos": 1.4039702725617254e+17,
|
|
"train_batch_size": 4,
|
|
"trial_name": null,
|
|
"trial_params": null
|
|
}
|