Files
qwen2.5-0.5b-sft-IT/checkpoint-500/trainer_state.json
ModelHub XC ca60016a55 初始化项目,由ModelHub XC社区提供模型
Model: adityabanerjee13/qwen2.5-0.5b-sft-IT
Source: Original Platform
2026-09-13 17:18:17 +08:00

976 lines
31 KiB
JSON

{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 2.526582278481013,
"eval_steps": 50,
"global_step": 500,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0,
"eval_indic_sft_mini_val_loss": 1.127184510231018,
"eval_indic_sft_mini_val_runtime": 156.0033,
"eval_indic_sft_mini_val_samples_per_second": 12.307,
"eval_indic_sft_mini_val_steps_per_second": 3.077,
"memory/device_reserved (GiB)": 24.22,
"memory/max_active (GiB)": 24.14,
"memory/max_allocated (GiB)": 24.14,
"step": 0
},
{
"epoch": 0,
"eval_tulu_sft_mini_val_loss": 2.330348491668701,
"eval_tulu_sft_mini_val_runtime": 91.8069,
"eval_tulu_sft_mini_val_samples_per_second": 12.548,
"eval_tulu_sft_mini_val_steps_per_second": 3.137,
"memory/device_reserved (GiB)": 24.22,
"memory/max_active (GiB)": 24.14,
"memory/max_allocated (GiB)": 24.14,
"step": 0
},
{
"epoch": 0.05063291139240506,
"grad_norm": 1.9375,
"learning_rate": 1.0588235294117648e-05,
"loss": 1.2239381790161132,
"memory/device_reserved (GiB)": 37.02,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 3.40055,
"step": 10,
"tokens/total": 1310720,
"tokens/trainable": 1268231
},
{
"epoch": 0.10126582278481013,
"grad_norm": 1.640625,
"learning_rate": 1.9999400896826965e-05,
"loss": 1.1529170989990234,
"memory/device_reserved (GiB)": 37.02,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 3.16742,
"step": 20,
"tokens/total": 2621440,
"tokens/train_per_sec_per_gpu": 17988.68,
"tokens/trainable": 2535045
},
{
"epoch": 0.1518987341772152,
"grad_norm": 1.4453125,
"learning_rate": 1.9978439822224228e-05,
"loss": 1.1212153434753418,
"memory/device_reserved (GiB)": 37.02,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 3.06858,
"step": 30,
"tokens/total": 3932160,
"tokens/train_per_sec_per_gpu": 17958.15,
"tokens/trainable": 3804033
},
{
"epoch": 0.20253164556962025,
"grad_norm": 1.3828125,
"learning_rate": 1.9927595335238736e-05,
"loss": 1.106326198577881,
"memory/device_reserved (GiB)": 37.02,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 3.02323,
"step": 40,
"tokens/total": 5242880,
"tokens/train_per_sec_per_gpu": 17929.18,
"tokens/trainable": 5070883
},
{
"epoch": 0.25316455696202533,
"grad_norm": 1.421875,
"learning_rate": 1.984701970484229e-05,
"loss": 1.0841044425964355,
"memory/device_reserved (GiB)": 37.02,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.95679,
"step": 50,
"tokens/total": 6553600,
"tokens/train_per_sec_per_gpu": 17943.82,
"tokens/trainable": 6341109
},
{
"epoch": 0.25316455696202533,
"eval_indic_sft_mini_val_loss": 1.0926785469055176,
"eval_indic_sft_mini_val_runtime": 157.8472,
"eval_indic_sft_mini_val_samples_per_second": 12.164,
"eval_indic_sft_mini_val_steps_per_second": 3.041,
"memory/device_reserved (GiB)": 37.02,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 50
},
{
"epoch": 0.25316455696202533,
"eval_tulu_sft_mini_val_loss": 2.3083696365356445,
"eval_tulu_sft_mini_val_runtime": 92.1987,
"eval_tulu_sft_mini_val_samples_per_second": 12.495,
"eval_tulu_sft_mini_val_steps_per_second": 3.124,
"memory/device_reserved (GiB)": 26.58,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 50
},
{
"epoch": 0.3037974683544304,
"grad_norm": 1.34375,
"learning_rate": 1.9736954238777793e-05,
"loss": 1.1303999900817872,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 3.09689,
"step": 60,
"tokens/total": 7864320,
"tokens/train_per_sec_per_gpu": 3953.8,
"tokens/trainable": 7607732
},
{
"epoch": 0.35443037974683544,
"grad_norm": 1.4296875,
"learning_rate": 1.9597728560891266e-05,
"loss": 1.1227096557617187,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 3.07317,
"step": 70,
"tokens/total": 9175040,
"tokens/train_per_sec_per_gpu": 17981.63,
"tokens/trainable": 8877038
},
{
"epoch": 0.4050632911392405,
"grad_norm": 1.3828125,
"learning_rate": 1.9429759623974992e-05,
"loss": 1.109241771697998,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 3.03206,
"step": 80,
"tokens/total": 10485760,
"tokens/train_per_sec_per_gpu": 17934.18,
"tokens/trainable": 10145302
},
{
"epoch": 0.45569620253164556,
"grad_norm": 1.2578125,
"learning_rate": 1.9233550461078114e-05,
"loss": 1.071034049987793,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.9184,
"step": 90,
"tokens/total": 11796480,
"tokens/train_per_sec_per_gpu": 17952.48,
"tokens/trainable": 11415878
},
{
"epoch": 0.5063291139240507,
"grad_norm": 1.2421875,
"learning_rate": 1.900968867902419e-05,
"loss": 1.0816166877746582,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.94944,
"step": 100,
"tokens/total": 13107200,
"tokens/train_per_sec_per_gpu": 17943.02,
"tokens/trainable": 12684573
},
{
"epoch": 0.5063291139240507,
"eval_indic_sft_mini_val_loss": 1.0623797178268433,
"eval_indic_sft_mini_val_runtime": 157.7495,
"eval_indic_sft_mini_val_samples_per_second": 12.171,
"eval_indic_sft_mini_val_steps_per_second": 3.043,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 100
},
{
"epoch": 0.5063291139240507,
"eval_tulu_sft_mini_val_loss": 2.2967746257781982,
"eval_tulu_sft_mini_val_runtime": 92.342,
"eval_tulu_sft_mini_val_samples_per_second": 12.475,
"eval_tulu_sft_mini_val_steps_per_second": 3.119,
"memory/device_reserved (GiB)": 26.58,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 100
},
{
"epoch": 0.5569620253164557,
"grad_norm": 1.2421875,
"learning_rate": 1.8758844698647457e-05,
"loss": 1.106839370727539,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 3.02478,
"step": 110,
"tokens/total": 14417920,
"tokens/train_per_sec_per_gpu": 3955.64,
"tokens/trainable": 13951995
},
{
"epoch": 0.6075949367088608,
"grad_norm": 1.8984375,
"learning_rate": 1.848176974701775e-05,
"loss": 1.0786801338195802,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.9408,
"step": 120,
"tokens/total": 15728640,
"tokens/train_per_sec_per_gpu": 17975.05,
"tokens/trainable": 15221447
},
{
"epoch": 0.6582278481012658,
"grad_norm": 1.28125,
"learning_rate": 1.8179293607667177e-05,
"loss": 1.0739567756652832,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.92694,
"step": 130,
"tokens/total": 17039360,
"tokens/train_per_sec_per_gpu": 17954.31,
"tokens/trainable": 16490903
},
{
"epoch": 0.7088607594936709,
"grad_norm": 1.2890625,
"learning_rate": 1.7852322135555946e-05,
"loss": 1.0759190559387206,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.93269,
"step": 140,
"tokens/total": 18350080,
"tokens/train_per_sec_per_gpu": 17887.36,
"tokens/trainable": 17757184
},
{
"epoch": 0.759493670886076,
"grad_norm": 1.3828125,
"learning_rate": 1.7501834544219697e-05,
"loss": 1.0366504669189454,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.81976,
"step": 150,
"tokens/total": 19660800,
"tokens/train_per_sec_per_gpu": 17911.77,
"tokens/trainable": 19024962
},
{
"epoch": 0.759493670886076,
"eval_indic_sft_mini_val_loss": 1.0453351736068726,
"eval_indic_sft_mini_val_runtime": 157.7623,
"eval_indic_sft_mini_val_samples_per_second": 12.17,
"eval_indic_sft_mini_val_steps_per_second": 3.043,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 150
},
{
"epoch": 0.759493670886076,
"eval_tulu_sft_mini_val_loss": 2.2883615493774414,
"eval_tulu_sft_mini_val_runtime": 92.4616,
"eval_tulu_sft_mini_val_samples_per_second": 12.459,
"eval_tulu_sft_mini_val_steps_per_second": 3.115,
"memory/device_reserved (GiB)": 26.58,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 150
},
{
"epoch": 0.810126582278481,
"grad_norm": 1.328125,
"learning_rate": 1.7128880473222688e-05,
"loss": 1.0812637329101562,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.9484,
"step": 160,
"tokens/total": 20971520,
"tokens/train_per_sec_per_gpu": 3951.59,
"tokens/trainable": 20291324
},
{
"epoch": 0.8607594936708861,
"grad_norm": 1.328125,
"learning_rate": 1.6734576844699234e-05,
"loss": 1.0667606353759767,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.90595,
"step": 170,
"tokens/total": 22282240,
"tokens/train_per_sec_per_gpu": 17990.97,
"tokens/trainable": 21559984
},
{
"epoch": 0.9113924050632911,
"grad_norm": 1.3515625,
"learning_rate": 1.6320104518397473e-05,
"loss": 1.067934513092041,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.90936,
"step": 180,
"tokens/total": 23592960,
"tokens/train_per_sec_per_gpu": 17944.73,
"tokens/trainable": 22828372
},
{
"epoch": 0.9620253164556962,
"grad_norm": 1.296875,
"learning_rate": 1.588670475524283e-05,
"loss": 1.04850492477417,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.85338,
"step": 190,
"tokens/total": 24903680,
"tokens/train_per_sec_per_gpu": 17890.69,
"tokens/trainable": 24094636
},
{
"epoch": 1.010126582278481,
"grad_norm": 1.3046875,
"learning_rate": 1.5435675500012212e-05,
"loss": 1.0491924285888672,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.85534,
"step": 200,
"tokens/total": 26136576,
"tokens/train_per_sec_per_gpu": 17465.18,
"tokens/trainable": 25286404
},
{
"epoch": 1.010126582278481,
"eval_indic_sft_mini_val_loss": 1.0354256629943848,
"eval_indic_sft_mini_val_runtime": 157.741,
"eval_indic_sft_mini_val_samples_per_second": 12.172,
"eval_indic_sft_mini_val_steps_per_second": 3.043,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 200
},
{
"epoch": 1.010126582278481,
"eval_tulu_sft_mini_val_loss": 2.291062593460083,
"eval_tulu_sft_mini_val_runtime": 92.344,
"eval_tulu_sft_mini_val_samples_per_second": 12.475,
"eval_tulu_sft_mini_val_steps_per_second": 3.119,
"memory/device_reserved (GiB)": 26.58,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 200
},
{
"epoch": 1.0607594936708862,
"grad_norm": 1.375,
"learning_rate": 1.4968367494251486e-05,
"loss": 1.0487144470214844,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.85398,
"step": 210,
"tokens/total": 27447296,
"tokens/train_per_sec_per_gpu": 3957.26,
"tokens/trainable": 26554084
},
{
"epoch": 1.111392405063291,
"grad_norm": 1.3671875,
"learning_rate": 1.4486180231077278e-05,
"loss": 1.0355692863464356,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.81671,
"step": 220,
"tokens/total": 28758016,
"tokens/train_per_sec_per_gpu": 17979.8,
"tokens/trainable": 27823280
},
{
"epoch": 1.1620253164556962,
"grad_norm": 1.265625,
"learning_rate": 1.3990557763977694e-05,
"loss": 1.0380287170410156,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.82365,
"step": 230,
"tokens/total": 30068736,
"tokens/train_per_sec_per_gpu": 17964.75,
"tokens/trainable": 29092706
},
{
"epoch": 1.2126582278481013,
"grad_norm": 1.2734375,
"learning_rate": 1.3482984382163713e-05,
"loss": 1.036821174621582,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.82024,
"step": 240,
"tokens/total": 31379456,
"tokens/train_per_sec_per_gpu": 17930.26,
"tokens/trainable": 30360732
},
{
"epoch": 1.2632911392405064,
"grad_norm": 1.2265625,
"learning_rate": 1.2964980165422701e-05,
"loss": 0.995778751373291,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.70683,
"step": 250,
"tokens/total": 32690176,
"tokens/train_per_sec_per_gpu": 17927.94,
"tokens/trainable": 31631352
},
{
"epoch": 1.2632911392405064,
"eval_indic_sft_mini_val_loss": 1.027944803237915,
"eval_indic_sft_mini_val_runtime": 157.8775,
"eval_indic_sft_mini_val_samples_per_second": 12.161,
"eval_indic_sft_mini_val_steps_per_second": 3.04,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 250
},
{
"epoch": 1.2632911392405064,
"eval_tulu_sft_mini_val_loss": 2.2984113693237305,
"eval_tulu_sft_mini_val_runtime": 92.3048,
"eval_tulu_sft_mini_val_samples_per_second": 12.48,
"eval_tulu_sft_mini_val_steps_per_second": 3.12,
"memory/device_reserved (GiB)": 26.58,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 250
},
{
"epoch": 1.3139240506329113,
"grad_norm": 1.2109375,
"learning_rate": 1.2438096431786408e-05,
"loss": 1.0438777923583984,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.84021,
"step": 260,
"tokens/total": 34000896,
"tokens/train_per_sec_per_gpu": 3955.89,
"tokens/trainable": 32899468
},
{
"epoch": 1.3645569620253164,
"grad_norm": 1.28125,
"learning_rate": 1.1903911091646684e-05,
"loss": 1.0240283012390137,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.78439,
"step": 270,
"tokens/total": 35311616,
"tokens/train_per_sec_per_gpu": 17961.08,
"tokens/trainable": 34168384
},
{
"epoch": 1.4151898734177215,
"grad_norm": 1.234375,
"learning_rate": 1.1364023922232503e-05,
"loss": 1.031261920928955,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.8046,
"step": 280,
"tokens/total": 36622336,
"tokens/train_per_sec_per_gpu": 17939.56,
"tokens/trainable": 35436704
},
{
"epoch": 1.4658227848101266,
"grad_norm": 1.3359375,
"learning_rate": 1.0820051776600175e-05,
"loss": 1.0369884490966796,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.82071,
"step": 290,
"tokens/total": 37933056,
"tokens/train_per_sec_per_gpu": 17936.98,
"tokens/trainable": 36705368
},
{
"epoch": 1.5164556962025317,
"grad_norm": 1.1875,
"learning_rate": 1.0273623741484924e-05,
"loss": 1.0238310813903808,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.78384,
"step": 300,
"tokens/total": 39243776,
"tokens/train_per_sec_per_gpu": 17926.64,
"tokens/trainable": 37973400
},
{
"epoch": 1.5164556962025317,
"eval_indic_sft_mini_val_loss": 1.0224663019180298,
"eval_indic_sft_mini_val_runtime": 157.771,
"eval_indic_sft_mini_val_samples_per_second": 12.17,
"eval_indic_sft_mini_val_steps_per_second": 3.042,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 300
},
{
"epoch": 1.5164556962025317,
"eval_tulu_sft_mini_val_loss": 2.295070171356201,
"eval_tulu_sft_mini_val_runtime": 92.349,
"eval_tulu_sft_mini_val_samples_per_second": 12.474,
"eval_tulu_sft_mini_val_steps_per_second": 3.119,
"memory/device_reserved (GiB)": 26.58,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 300
},
{
"epoch": 1.5670886075949366,
"grad_norm": 1.2421875,
"learning_rate": 9.726376258515077e-06,
"loss": 1.0644823074340821,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.89934,
"step": 310,
"tokens/total": 40554496,
"tokens/train_per_sec_per_gpu": 3956.31,
"tokens/trainable": 39240980
},
{
"epoch": 1.6177215189873417,
"grad_norm": 1.234375,
"learning_rate": 9.179948223399828e-06,
"loss": 1.0305088996887206,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.80249,
"step": 320,
"tokens/total": 41865216,
"tokens/train_per_sec_per_gpu": 17965.06,
"tokens/trainable": 40509632
},
{
"epoch": 1.6683544303797468,
"grad_norm": 1.265625,
"learning_rate": 8.6359760777675e-06,
"loss": 1.0250298500061035,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.78718,
"step": 330,
"tokens/total": 43175936,
"tokens/train_per_sec_per_gpu": 17953.88,
"tokens/trainable": 41777768
},
{
"epoch": 1.7189873417721517,
"grad_norm": 1.2265625,
"learning_rate": 8.096088908353316e-06,
"loss": 1.0106207847595214,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.74731,
"step": 340,
"tokens/total": 44486656,
"tokens/train_per_sec_per_gpu": 17933.81,
"tokens/trainable": 43045456
},
{
"epoch": 1.769620253164557,
"grad_norm": 1.2578125,
"learning_rate": 7.561903568213595e-06,
"loss": 0.9956131935119629,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.70638,
"step": 350,
"tokens/total": 45797376,
"tokens/train_per_sec_per_gpu": 17927.21,
"tokens/trainable": 44311528
},
{
"epoch": 1.769620253164557,
"eval_indic_sft_mini_val_loss": 1.0202709436416626,
"eval_indic_sft_mini_val_runtime": 157.7597,
"eval_indic_sft_mini_val_samples_per_second": 12.17,
"eval_indic_sft_mini_val_steps_per_second": 3.043,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 350
},
{
"epoch": 1.769620253164557,
"eval_tulu_sft_mini_val_loss": 2.2989728450775146,
"eval_tulu_sft_mini_val_runtime": 92.4784,
"eval_tulu_sft_mini_val_samples_per_second": 12.457,
"eval_tulu_sft_mini_val_steps_per_second": 3.114,
"memory/device_reserved (GiB)": 26.58,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 350
},
{
"epoch": 1.820253164556962,
"grad_norm": 1.2734375,
"learning_rate": 7.035019834577301e-06,
"loss": 1.0437658309936524,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.83989,
"step": 360,
"tokens/total": 47108096,
"tokens/train_per_sec_per_gpu": 3952.58,
"tokens/trainable": 45578288
},
{
"epoch": 1.870886075949367,
"grad_norm": 1.2265625,
"learning_rate": 6.517015617836292e-06,
"loss": 1.0040513038635255,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.72932,
"step": 370,
"tokens/total": 48418816,
"tokens/train_per_sec_per_gpu": 17939.3,
"tokens/trainable": 46846324
},
{
"epoch": 1.9215189873417722,
"grad_norm": 1.234375,
"learning_rate": 6.009442236022307e-06,
"loss": 1.0151902198791505,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.75989,
"step": 380,
"tokens/total": 49729536,
"tokens/train_per_sec_per_gpu": 17915.04,
"tokens/trainable": 48113428
},
{
"epoch": 1.972151898734177,
"grad_norm": 1.3125,
"learning_rate": 5.513819768922723e-06,
"loss": 0.997506046295166,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.71151,
"step": 390,
"tokens/total": 51040256,
"tokens/train_per_sec_per_gpu": 17963.73,
"tokens/trainable": 49382776
},
{
"epoch": 2.020253164556962,
"grad_norm": 1.2421875,
"learning_rate": 5.031632505748516e-06,
"loss": 1.0011167526245117,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.72132,
"step": 400,
"tokens/total": 52273152,
"tokens/train_per_sec_per_gpu": 17458.66,
"tokens/trainable": 50572704
},
{
"epoch": 2.020253164556962,
"eval_indic_sft_mini_val_loss": 1.0187087059020996,
"eval_indic_sft_mini_val_runtime": 158.0242,
"eval_indic_sft_mini_val_samples_per_second": 12.15,
"eval_indic_sft_mini_val_steps_per_second": 3.038,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 400
},
{
"epoch": 2.020253164556962,
"eval_tulu_sft_mini_val_loss": 2.2966930866241455,
"eval_tulu_sft_mini_val_runtime": 92.3222,
"eval_tulu_sft_mini_val_samples_per_second": 12.478,
"eval_tulu_sft_mini_val_steps_per_second": 3.12,
"memory/device_reserved (GiB)": 26.58,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 400
},
{
"epoch": 2.070886075949367,
"grad_norm": 1.2421875,
"learning_rate": 4.56432449998779e-06,
"loss": 1.0103323936462403,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.74651,
"step": 410,
"tokens/total": 53583872,
"tokens/train_per_sec_per_gpu": 3952.63,
"tokens/trainable": 51839796
},
{
"epoch": 2.1215189873417724,
"grad_norm": 1.234375,
"learning_rate": 4.113295244757171e-06,
"loss": 1.0133058547973632,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.75469,
"step": 420,
"tokens/total": 54894592,
"tokens/train_per_sec_per_gpu": 17984.79,
"tokens/trainable": 53108816
},
{
"epoch": 2.1721518987341772,
"grad_norm": 1.21875,
"learning_rate": 3.679895481602529e-06,
"loss": 1.0214984893798829,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.77735,
"step": 430,
"tokens/total": 56205312,
"tokens/train_per_sec_per_gpu": 17959.57,
"tokens/trainable": 54378168
},
{
"epoch": 2.222784810126582,
"grad_norm": 1.2109375,
"learning_rate": 3.2654231553007665e-06,
"loss": 0.9840593338012695,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.67529,
"step": 440,
"tokens/total": 57516032,
"tokens/train_per_sec_per_gpu": 17930.45,
"tokens/trainable": 55645416
},
{
"epoch": 2.2734177215189875,
"grad_norm": 1.1953125,
"learning_rate": 2.871119526777315e-06,
"loss": 1.028823184967041,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.79777,
"step": 450,
"tokens/total": 58826752,
"tokens/train_per_sec_per_gpu": 17944.17,
"tokens/trainable": 56914256
},
{
"epoch": 2.2734177215189875,
"eval_indic_sft_mini_val_loss": 1.0182074308395386,
"eval_indic_sft_mini_val_runtime": 157.9039,
"eval_indic_sft_mini_val_samples_per_second": 12.159,
"eval_indic_sft_mini_val_steps_per_second": 3.04,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 450
},
{
"epoch": 2.2734177215189875,
"eval_tulu_sft_mini_val_loss": 2.297419309616089,
"eval_tulu_sft_mini_val_runtime": 92.4484,
"eval_tulu_sft_mini_val_samples_per_second": 12.461,
"eval_tulu_sft_mini_val_steps_per_second": 3.115,
"memory/device_reserved (GiB)": 26.59,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 450
},
{
"epoch": 2.3240506329113924,
"grad_norm": 1.21875,
"learning_rate": 2.4981654557803026e-06,
"loss": 1.0112363815307617,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.749,
"step": 460,
"tokens/total": 60137472,
"tokens/train_per_sec_per_gpu": 3954.28,
"tokens/trainable": 58182604
},
{
"epoch": 2.3746835443037977,
"grad_norm": 1.265625,
"learning_rate": 2.1476778644440553e-06,
"loss": 0.9997093200683593,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.71749,
"step": 470,
"tokens/total": 61448192,
"tokens/train_per_sec_per_gpu": 17956.34,
"tokens/trainable": 59450748
},
{
"epoch": 2.4253164556962026,
"grad_norm": 1.21875,
"learning_rate": 1.820706392332824e-06,
"loss": 1.0180435180664062,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.76777,
"step": 480,
"tokens/total": 62758912,
"tokens/train_per_sec_per_gpu": 17939.84,
"tokens/trainable": 60720528
},
{
"epoch": 2.4759493670886075,
"grad_norm": 1.234375,
"learning_rate": 1.518230252982248e-06,
"loss": 1.040645217895508,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.83104,
"step": 490,
"tokens/total": 64069632,
"tokens/train_per_sec_per_gpu": 17931.8,
"tokens/trainable": 61988728
},
{
"epoch": 2.526582278481013,
"grad_norm": 1.171875,
"learning_rate": 1.2411553013525457e-06,
"loss": 1.0239150047302246,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.78407,
"step": 500,
"tokens/total": 65380352,
"tokens/train_per_sec_per_gpu": 17916.35,
"tokens/trainable": 63254896
},
{
"epoch": 2.526582278481013,
"eval_indic_sft_mini_val_loss": 1.018338680267334,
"eval_indic_sft_mini_val_runtime": 157.6995,
"eval_indic_sft_mini_val_samples_per_second": 12.175,
"eval_indic_sft_mini_val_steps_per_second": 3.044,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 500
},
{
"epoch": 2.526582278481013,
"eval_tulu_sft_mini_val_loss": 2.299147605895996,
"eval_tulu_sft_mini_val_runtime": 92.1488,
"eval_tulu_sft_mini_val_samples_per_second": 12.502,
"eval_tulu_sft_mini_val_steps_per_second": 3.125,
"memory/device_reserved (GiB)": 26.59,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 500
}
],
"logging_steps": 10,
"max_steps": 591,
"num_input_tokens_seen": 0,
"num_train_epochs": 3,
"save_steps": 500,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 1.4039702725617254e+17,
"train_batch_size": 4,
"trial_name": null,
"trial_params": null
}