{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 2.9873417721518987, "eval_steps": 50, "global_step": 591, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0, "eval_indic_sft_mini_val_loss": 1.127184510231018, "eval_indic_sft_mini_val_runtime": 156.0033, "eval_indic_sft_mini_val_samples_per_second": 12.307, "eval_indic_sft_mini_val_steps_per_second": 3.077, "memory/device_reserved (GiB)": 24.22, "memory/max_active (GiB)": 24.14, "memory/max_allocated (GiB)": 24.14, "step": 0 }, { "epoch": 0, "eval_tulu_sft_mini_val_loss": 2.330348491668701, "eval_tulu_sft_mini_val_runtime": 91.8069, "eval_tulu_sft_mini_val_samples_per_second": 12.548, "eval_tulu_sft_mini_val_steps_per_second": 3.137, "memory/device_reserved (GiB)": 24.22, "memory/max_active (GiB)": 24.14, "memory/max_allocated (GiB)": 24.14, "step": 0 }, { "epoch": 0.05063291139240506, "grad_norm": 1.9375, "learning_rate": 1.0588235294117648e-05, "loss": 1.2239381790161132, "memory/device_reserved (GiB)": 37.02, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 3.40055, "step": 10, "tokens/total": 1310720, "tokens/trainable": 1268231 }, { "epoch": 0.10126582278481013, "grad_norm": 1.640625, "learning_rate": 1.9999400896826965e-05, "loss": 1.1529170989990234, "memory/device_reserved (GiB)": 37.02, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 3.16742, "step": 20, "tokens/total": 2621440, "tokens/train_per_sec_per_gpu": 17988.68, "tokens/trainable": 2535045 }, { "epoch": 0.1518987341772152, "grad_norm": 1.4453125, "learning_rate": 1.9978439822224228e-05, "loss": 1.1212153434753418, "memory/device_reserved (GiB)": 37.02, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 3.06858, "step": 30, "tokens/total": 3932160, "tokens/train_per_sec_per_gpu": 17958.15, "tokens/trainable": 3804033 }, { "epoch": 0.20253164556962025, "grad_norm": 1.3828125, "learning_rate": 1.9927595335238736e-05, "loss": 1.106326198577881, "memory/device_reserved (GiB)": 37.02, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 3.02323, "step": 40, "tokens/total": 5242880, "tokens/train_per_sec_per_gpu": 17929.18, "tokens/trainable": 5070883 }, { "epoch": 0.25316455696202533, "grad_norm": 1.421875, "learning_rate": 1.984701970484229e-05, "loss": 1.0841044425964355, "memory/device_reserved (GiB)": 37.02, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.95679, "step": 50, "tokens/total": 6553600, "tokens/train_per_sec_per_gpu": 17943.82, "tokens/trainable": 6341109 }, { "epoch": 0.25316455696202533, "eval_indic_sft_mini_val_loss": 1.0926785469055176, "eval_indic_sft_mini_val_runtime": 157.8472, "eval_indic_sft_mini_val_samples_per_second": 12.164, "eval_indic_sft_mini_val_steps_per_second": 3.041, "memory/device_reserved (GiB)": 37.02, "memory/max_active (GiB)": 25.99, "memory/max_allocated (GiB)": 25.99, "step": 50 }, { "epoch": 0.25316455696202533, "eval_tulu_sft_mini_val_loss": 2.3083696365356445, "eval_tulu_sft_mini_val_runtime": 92.1987, "eval_tulu_sft_mini_val_samples_per_second": 12.495, "eval_tulu_sft_mini_val_steps_per_second": 3.124, "memory/device_reserved (GiB)": 26.58, "memory/max_active (GiB)": 25.99, "memory/max_allocated (GiB)": 25.99, "step": 50 }, { "epoch": 0.3037974683544304, "grad_norm": 1.34375, "learning_rate": 1.9736954238777793e-05, "loss": 1.1303999900817872, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 3.09689, "step": 60, "tokens/total": 7864320, "tokens/train_per_sec_per_gpu": 3953.8, "tokens/trainable": 7607732 }, { "epoch": 0.35443037974683544, "grad_norm": 1.4296875, "learning_rate": 1.9597728560891266e-05, "loss": 1.1227096557617187, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 3.07317, "step": 70, "tokens/total": 9175040, "tokens/train_per_sec_per_gpu": 17981.63, "tokens/trainable": 8877038 }, { "epoch": 0.4050632911392405, "grad_norm": 1.3828125, "learning_rate": 1.9429759623974992e-05, "loss": 1.109241771697998, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 3.03206, "step": 80, "tokens/total": 10485760, "tokens/train_per_sec_per_gpu": 17934.18, "tokens/trainable": 10145302 }, { "epoch": 0.45569620253164556, "grad_norm": 1.2578125, "learning_rate": 1.9233550461078114e-05, "loss": 1.071034049987793, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.9184, "step": 90, "tokens/total": 11796480, "tokens/train_per_sec_per_gpu": 17952.48, "tokens/trainable": 11415878 }, { "epoch": 0.5063291139240507, "grad_norm": 1.2421875, "learning_rate": 1.900968867902419e-05, "loss": 1.0816166877746582, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.94944, "step": 100, "tokens/total": 13107200, "tokens/train_per_sec_per_gpu": 17943.02, "tokens/trainable": 12684573 }, { "epoch": 0.5063291139240507, "eval_indic_sft_mini_val_loss": 1.0623797178268433, "eval_indic_sft_mini_val_runtime": 157.7495, "eval_indic_sft_mini_val_samples_per_second": 12.171, "eval_indic_sft_mini_val_steps_per_second": 3.043, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 25.99, "memory/max_allocated (GiB)": 25.99, "step": 100 }, { "epoch": 0.5063291139240507, "eval_tulu_sft_mini_val_loss": 2.2967746257781982, "eval_tulu_sft_mini_val_runtime": 92.342, "eval_tulu_sft_mini_val_samples_per_second": 12.475, "eval_tulu_sft_mini_val_steps_per_second": 3.119, "memory/device_reserved (GiB)": 26.58, "memory/max_active (GiB)": 25.99, "memory/max_allocated (GiB)": 25.99, "step": 100 }, { "epoch": 0.5569620253164557, "grad_norm": 1.2421875, "learning_rate": 1.8758844698647457e-05, "loss": 1.106839370727539, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 3.02478, "step": 110, "tokens/total": 14417920, "tokens/train_per_sec_per_gpu": 3955.64, "tokens/trainable": 13951995 }, { "epoch": 0.6075949367088608, "grad_norm": 1.8984375, "learning_rate": 1.848176974701775e-05, "loss": 1.0786801338195802, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.9408, "step": 120, "tokens/total": 15728640, "tokens/train_per_sec_per_gpu": 17975.05, "tokens/trainable": 15221447 }, { "epoch": 0.6582278481012658, "grad_norm": 1.28125, "learning_rate": 1.8179293607667177e-05, "loss": 1.0739567756652832, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.92694, "step": 130, "tokens/total": 17039360, "tokens/train_per_sec_per_gpu": 17954.31, "tokens/trainable": 16490903 }, { "epoch": 0.7088607594936709, "grad_norm": 1.2890625, "learning_rate": 1.7852322135555946e-05, "loss": 1.0759190559387206, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.93269, "step": 140, "tokens/total": 18350080, "tokens/train_per_sec_per_gpu": 17887.36, "tokens/trainable": 17757184 }, { "epoch": 0.759493670886076, "grad_norm": 1.3828125, "learning_rate": 1.7501834544219697e-05, "loss": 1.0366504669189454, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.81976, "step": 150, "tokens/total": 19660800, "tokens/train_per_sec_per_gpu": 17911.77, "tokens/trainable": 19024962 }, { "epoch": 0.759493670886076, "eval_indic_sft_mini_val_loss": 1.0453351736068726, "eval_indic_sft_mini_val_runtime": 157.7623, "eval_indic_sft_mini_val_samples_per_second": 12.17, "eval_indic_sft_mini_val_steps_per_second": 3.043, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 25.99, "memory/max_allocated (GiB)": 25.99, "step": 150 }, { "epoch": 0.759493670886076, "eval_tulu_sft_mini_val_loss": 2.2883615493774414, "eval_tulu_sft_mini_val_runtime": 92.4616, "eval_tulu_sft_mini_val_samples_per_second": 12.459, "eval_tulu_sft_mini_val_steps_per_second": 3.115, "memory/device_reserved (GiB)": 26.58, "memory/max_active (GiB)": 25.99, "memory/max_allocated (GiB)": 25.99, "step": 150 }, { "epoch": 0.810126582278481, "grad_norm": 1.328125, "learning_rate": 1.7128880473222688e-05, "loss": 1.0812637329101562, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.9484, "step": 160, "tokens/total": 20971520, "tokens/train_per_sec_per_gpu": 3951.59, "tokens/trainable": 20291324 }, { "epoch": 0.8607594936708861, "grad_norm": 1.328125, "learning_rate": 1.6734576844699234e-05, "loss": 1.0667606353759767, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.90595, "step": 170, "tokens/total": 22282240, "tokens/train_per_sec_per_gpu": 17990.97, "tokens/trainable": 21559984 }, { "epoch": 0.9113924050632911, "grad_norm": 1.3515625, "learning_rate": 1.6320104518397473e-05, "loss": 1.067934513092041, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.90936, "step": 180, "tokens/total": 23592960, "tokens/train_per_sec_per_gpu": 17944.73, "tokens/trainable": 22828372 }, { "epoch": 0.9620253164556962, "grad_norm": 1.296875, "learning_rate": 1.588670475524283e-05, "loss": 1.04850492477417, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.85338, "step": 190, "tokens/total": 24903680, "tokens/train_per_sec_per_gpu": 17890.69, "tokens/trainable": 24094636 }, { "epoch": 1.010126582278481, "grad_norm": 1.3046875, "learning_rate": 1.5435675500012212e-05, "loss": 1.0491924285888672, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.85534, "step": 200, "tokens/total": 26136576, "tokens/train_per_sec_per_gpu": 17465.18, "tokens/trainable": 25286404 }, { "epoch": 1.010126582278481, "eval_indic_sft_mini_val_loss": 1.0354256629943848, "eval_indic_sft_mini_val_runtime": 157.741, "eval_indic_sft_mini_val_samples_per_second": 12.172, "eval_indic_sft_mini_val_steps_per_second": 3.043, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 25.99, "memory/max_allocated (GiB)": 25.99, "step": 200 }, { "epoch": 1.010126582278481, "eval_tulu_sft_mini_val_loss": 2.291062593460083, "eval_tulu_sft_mini_val_runtime": 92.344, "eval_tulu_sft_mini_val_samples_per_second": 12.475, "eval_tulu_sft_mini_val_steps_per_second": 3.119, "memory/device_reserved (GiB)": 26.58, "memory/max_active (GiB)": 25.99, "memory/max_allocated (GiB)": 25.99, "step": 200 }, { "epoch": 1.0607594936708862, "grad_norm": 1.375, "learning_rate": 1.4968367494251486e-05, "loss": 1.0487144470214844, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.85398, "step": 210, "tokens/total": 27447296, "tokens/train_per_sec_per_gpu": 3957.26, "tokens/trainable": 26554084 }, { "epoch": 1.111392405063291, "grad_norm": 1.3671875, "learning_rate": 1.4486180231077278e-05, "loss": 1.0355692863464356, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.81671, "step": 220, "tokens/total": 28758016, "tokens/train_per_sec_per_gpu": 17979.8, "tokens/trainable": 27823280 }, { "epoch": 1.1620253164556962, "grad_norm": 1.265625, "learning_rate": 1.3990557763977694e-05, "loss": 1.0380287170410156, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.82365, "step": 230, "tokens/total": 30068736, "tokens/train_per_sec_per_gpu": 17964.75, "tokens/trainable": 29092706 }, { "epoch": 1.2126582278481013, "grad_norm": 1.2734375, "learning_rate": 1.3482984382163713e-05, "loss": 1.036821174621582, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.82024, "step": 240, "tokens/total": 31379456, "tokens/train_per_sec_per_gpu": 17930.26, "tokens/trainable": 30360732 }, { "epoch": 1.2632911392405064, "grad_norm": 1.2265625, "learning_rate": 1.2964980165422701e-05, "loss": 0.995778751373291, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.70683, "step": 250, "tokens/total": 32690176, "tokens/train_per_sec_per_gpu": 17927.94, "tokens/trainable": 31631352 }, { "epoch": 1.2632911392405064, "eval_indic_sft_mini_val_loss": 1.027944803237915, "eval_indic_sft_mini_val_runtime": 157.8775, "eval_indic_sft_mini_val_samples_per_second": 12.161, "eval_indic_sft_mini_val_steps_per_second": 3.04, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 25.99, "memory/max_allocated (GiB)": 25.99, "step": 250 }, { "epoch": 1.2632911392405064, "eval_tulu_sft_mini_val_loss": 2.2984113693237305, "eval_tulu_sft_mini_val_runtime": 92.3048, "eval_tulu_sft_mini_val_samples_per_second": 12.48, "eval_tulu_sft_mini_val_steps_per_second": 3.12, "memory/device_reserved (GiB)": 26.58, "memory/max_active (GiB)": 25.99, "memory/max_allocated (GiB)": 25.99, "step": 250 }, { "epoch": 1.3139240506329113, "grad_norm": 1.2109375, "learning_rate": 1.2438096431786408e-05, "loss": 1.0438777923583984, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.84021, "step": 260, "tokens/total": 34000896, "tokens/train_per_sec_per_gpu": 3955.89, "tokens/trainable": 32899468 }, { "epoch": 1.3645569620253164, "grad_norm": 1.28125, "learning_rate": 1.1903911091646684e-05, "loss": 1.0240283012390137, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.78439, "step": 270, "tokens/total": 35311616, "tokens/train_per_sec_per_gpu": 17961.08, "tokens/trainable": 34168384 }, { "epoch": 1.4151898734177215, "grad_norm": 1.234375, "learning_rate": 1.1364023922232503e-05, "loss": 1.031261920928955, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.8046, "step": 280, "tokens/total": 36622336, "tokens/train_per_sec_per_gpu": 17939.56, "tokens/trainable": 35436704 }, { "epoch": 1.4658227848101266, "grad_norm": 1.3359375, "learning_rate": 1.0820051776600175e-05, "loss": 1.0369884490966796, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.82071, "step": 290, "tokens/total": 37933056, "tokens/train_per_sec_per_gpu": 17936.98, "tokens/trainable": 36705368 }, { "epoch": 1.5164556962025317, "grad_norm": 1.1875, "learning_rate": 1.0273623741484924e-05, "loss": 1.0238310813903808, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.78384, "step": 300, "tokens/total": 39243776, "tokens/train_per_sec_per_gpu": 17926.64, "tokens/trainable": 37973400 }, { "epoch": 1.5164556962025317, "eval_indic_sft_mini_val_loss": 1.0224663019180298, "eval_indic_sft_mini_val_runtime": 157.771, "eval_indic_sft_mini_val_samples_per_second": 12.17, "eval_indic_sft_mini_val_steps_per_second": 3.042, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 25.99, "memory/max_allocated (GiB)": 25.99, "step": 300 }, { "epoch": 1.5164556962025317, "eval_tulu_sft_mini_val_loss": 2.295070171356201, "eval_tulu_sft_mini_val_runtime": 92.349, "eval_tulu_sft_mini_val_samples_per_second": 12.474, "eval_tulu_sft_mini_val_steps_per_second": 3.119, "memory/device_reserved (GiB)": 26.58, "memory/max_active (GiB)": 25.99, "memory/max_allocated (GiB)": 25.99, "step": 300 }, { "epoch": 1.5670886075949366, "grad_norm": 1.2421875, "learning_rate": 9.726376258515077e-06, "loss": 1.0644823074340821, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.89934, "step": 310, "tokens/total": 40554496, "tokens/train_per_sec_per_gpu": 3956.31, "tokens/trainable": 39240980 }, { "epoch": 1.6177215189873417, "grad_norm": 1.234375, "learning_rate": 9.179948223399828e-06, "loss": 1.0305088996887206, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.80249, "step": 320, "tokens/total": 41865216, "tokens/train_per_sec_per_gpu": 17965.06, "tokens/trainable": 40509632 }, { "epoch": 1.6683544303797468, "grad_norm": 1.265625, "learning_rate": 8.6359760777675e-06, "loss": 1.0250298500061035, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.78718, "step": 330, "tokens/total": 43175936, "tokens/train_per_sec_per_gpu": 17953.88, "tokens/trainable": 41777768 }, { "epoch": 1.7189873417721517, "grad_norm": 1.2265625, "learning_rate": 8.096088908353316e-06, "loss": 1.0106207847595214, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.74731, "step": 340, "tokens/total": 44486656, "tokens/train_per_sec_per_gpu": 17933.81, "tokens/trainable": 43045456 }, { "epoch": 1.769620253164557, "grad_norm": 1.2578125, "learning_rate": 7.561903568213595e-06, "loss": 0.9956131935119629, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.70638, "step": 350, "tokens/total": 45797376, "tokens/train_per_sec_per_gpu": 17927.21, "tokens/trainable": 44311528 }, { "epoch": 1.769620253164557, "eval_indic_sft_mini_val_loss": 1.0202709436416626, "eval_indic_sft_mini_val_runtime": 157.7597, "eval_indic_sft_mini_val_samples_per_second": 12.17, "eval_indic_sft_mini_val_steps_per_second": 3.043, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 25.99, "memory/max_allocated (GiB)": 25.99, "step": 350 }, { "epoch": 1.769620253164557, "eval_tulu_sft_mini_val_loss": 2.2989728450775146, "eval_tulu_sft_mini_val_runtime": 92.4784, "eval_tulu_sft_mini_val_samples_per_second": 12.457, "eval_tulu_sft_mini_val_steps_per_second": 3.114, "memory/device_reserved (GiB)": 26.58, "memory/max_active (GiB)": 25.99, "memory/max_allocated (GiB)": 25.99, "step": 350 }, { "epoch": 1.820253164556962, "grad_norm": 1.2734375, "learning_rate": 7.035019834577301e-06, "loss": 1.0437658309936524, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.83989, "step": 360, "tokens/total": 47108096, "tokens/train_per_sec_per_gpu": 3952.58, "tokens/trainable": 45578288 }, { "epoch": 1.870886075949367, "grad_norm": 1.2265625, "learning_rate": 6.517015617836292e-06, "loss": 1.0040513038635255, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.72932, "step": 370, "tokens/total": 48418816, "tokens/train_per_sec_per_gpu": 17939.3, "tokens/trainable": 46846324 }, { "epoch": 1.9215189873417722, "grad_norm": 1.234375, "learning_rate": 6.009442236022307e-06, "loss": 1.0151902198791505, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.75989, "step": 380, "tokens/total": 49729536, "tokens/train_per_sec_per_gpu": 17915.04, "tokens/trainable": 48113428 }, { "epoch": 1.972151898734177, "grad_norm": 1.3125, "learning_rate": 5.513819768922723e-06, "loss": 0.997506046295166, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.71151, "step": 390, "tokens/total": 51040256, "tokens/train_per_sec_per_gpu": 17963.73, "tokens/trainable": 49382776 }, { "epoch": 2.020253164556962, "grad_norm": 1.2421875, "learning_rate": 5.031632505748516e-06, "loss": 1.0011167526245117, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.72132, "step": 400, "tokens/total": 52273152, "tokens/train_per_sec_per_gpu": 17458.66, "tokens/trainable": 50572704 }, { "epoch": 2.020253164556962, "eval_indic_sft_mini_val_loss": 1.0187087059020996, "eval_indic_sft_mini_val_runtime": 158.0242, "eval_indic_sft_mini_val_samples_per_second": 12.15, "eval_indic_sft_mini_val_steps_per_second": 3.038, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 25.99, "memory/max_allocated (GiB)": 25.99, "step": 400 }, { "epoch": 2.020253164556962, "eval_tulu_sft_mini_val_loss": 2.2966930866241455, "eval_tulu_sft_mini_val_runtime": 92.3222, "eval_tulu_sft_mini_val_samples_per_second": 12.478, "eval_tulu_sft_mini_val_steps_per_second": 3.12, "memory/device_reserved (GiB)": 26.58, "memory/max_active (GiB)": 25.99, "memory/max_allocated (GiB)": 25.99, "step": 400 }, { "epoch": 2.070886075949367, "grad_norm": 1.2421875, "learning_rate": 4.56432449998779e-06, "loss": 1.0103323936462403, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.74651, "step": 410, "tokens/total": 53583872, "tokens/train_per_sec_per_gpu": 3952.63, "tokens/trainable": 51839796 }, { "epoch": 2.1215189873417724, "grad_norm": 1.234375, "learning_rate": 4.113295244757171e-06, "loss": 1.0133058547973632, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.75469, "step": 420, "tokens/total": 54894592, "tokens/train_per_sec_per_gpu": 17984.79, "tokens/trainable": 53108816 }, { "epoch": 2.1721518987341772, "grad_norm": 1.21875, "learning_rate": 3.679895481602529e-06, "loss": 1.0214984893798829, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.77735, "step": 430, "tokens/total": 56205312, "tokens/train_per_sec_per_gpu": 17959.57, "tokens/trainable": 54378168 }, { "epoch": 2.222784810126582, "grad_norm": 1.2109375, "learning_rate": 3.2654231553007665e-06, "loss": 0.9840593338012695, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.67529, "step": 440, "tokens/total": 57516032, "tokens/train_per_sec_per_gpu": 17930.45, "tokens/trainable": 55645416 }, { "epoch": 2.2734177215189875, "grad_norm": 1.1953125, "learning_rate": 2.871119526777315e-06, "loss": 1.028823184967041, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.79777, "step": 450, "tokens/total": 58826752, "tokens/train_per_sec_per_gpu": 17944.17, "tokens/trainable": 56914256 }, { "epoch": 2.2734177215189875, "eval_indic_sft_mini_val_loss": 1.0182074308395386, "eval_indic_sft_mini_val_runtime": 157.9039, "eval_indic_sft_mini_val_samples_per_second": 12.159, "eval_indic_sft_mini_val_steps_per_second": 3.04, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 25.99, "memory/max_allocated (GiB)": 25.99, "step": 450 }, { "epoch": 2.2734177215189875, "eval_tulu_sft_mini_val_loss": 2.297419309616089, "eval_tulu_sft_mini_val_runtime": 92.4484, "eval_tulu_sft_mini_val_samples_per_second": 12.461, "eval_tulu_sft_mini_val_steps_per_second": 3.115, "memory/device_reserved (GiB)": 26.59, "memory/max_active (GiB)": 25.99, "memory/max_allocated (GiB)": 25.99, "step": 450 }, { "epoch": 2.3240506329113924, "grad_norm": 1.21875, "learning_rate": 2.4981654557803026e-06, "loss": 1.0112363815307617, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.749, "step": 460, "tokens/total": 60137472, "tokens/train_per_sec_per_gpu": 3954.28, "tokens/trainable": 58182604 }, { "epoch": 2.3746835443037977, "grad_norm": 1.265625, "learning_rate": 2.1476778644440553e-06, "loss": 0.9997093200683593, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.71749, "step": 470, "tokens/total": 61448192, "tokens/train_per_sec_per_gpu": 17956.34, "tokens/trainable": 59450748 }, { "epoch": 2.4253164556962026, "grad_norm": 1.21875, "learning_rate": 1.820706392332824e-06, "loss": 1.0180435180664062, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.76777, "step": 480, "tokens/total": 62758912, "tokens/train_per_sec_per_gpu": 17939.84, "tokens/trainable": 60720528 }, { "epoch": 2.4759493670886075, "grad_norm": 1.234375, "learning_rate": 1.518230252982248e-06, "loss": 1.040645217895508, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.83104, "step": 490, "tokens/total": 64069632, "tokens/train_per_sec_per_gpu": 17931.8, "tokens/trainable": 61988728 }, { "epoch": 2.526582278481013, "grad_norm": 1.171875, "learning_rate": 1.2411553013525457e-06, "loss": 1.0239150047302246, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.78407, "step": 500, "tokens/total": 65380352, "tokens/train_per_sec_per_gpu": 17916.35, "tokens/trainable": 63254896 }, { "epoch": 2.526582278481013, "eval_indic_sft_mini_val_loss": 1.018338680267334, "eval_indic_sft_mini_val_runtime": 157.6995, "eval_indic_sft_mini_val_samples_per_second": 12.175, "eval_indic_sft_mini_val_steps_per_second": 3.044, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 25.99, "memory/max_allocated (GiB)": 25.99, "step": 500 }, { "epoch": 2.526582278481013, "eval_tulu_sft_mini_val_loss": 2.299147605895996, "eval_tulu_sft_mini_val_runtime": 92.1488, "eval_tulu_sft_mini_val_samples_per_second": 12.502, "eval_tulu_sft_mini_val_steps_per_second": 3.125, "memory/device_reserved (GiB)": 26.59, "memory/max_active (GiB)": 25.99, "memory/max_allocated (GiB)": 25.99, "step": 500 }, { "epoch": 2.5772151898734177, "grad_norm": 1.2109375, "learning_rate": 9.903113209758098e-07, "loss": 1.0059453964233398, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.73449, "step": 510, "tokens/total": 66691072, "tokens/train_per_sec_per_gpu": 3921.5, "tokens/trainable": 64525868 }, { "epoch": 2.6278481012658226, "grad_norm": 1.25, "learning_rate": 7.664495389218884e-07, "loss": 1.0043025016784668, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.73, "step": 520, "tokens/total": 68001792, "tokens/train_per_sec_per_gpu": 17995.95, "tokens/trainable": 65794468 }, { "epoch": 2.678481012658228, "grad_norm": 1.2265625, "learning_rate": 5.702403760250086e-07, "loss": 0.9985050201416016, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.71422, "step": 530, "tokens/total": 69312512, "tokens/train_per_sec_per_gpu": 17953.05, "tokens/trainable": 67063008 }, { "epoch": 2.729113924050633, "grad_norm": 1.234375, "learning_rate": 4.022714391087379e-07, "loss": 1.0013395309448243, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.72193, "step": 540, "tokens/total": 70623232, "tokens/train_per_sec_per_gpu": 17930.41, "tokens/trainable": 68331456 }, { "epoch": 2.779746835443038, "grad_norm": 1.2109375, "learning_rate": 2.6304576122221035e-07, "loss": 1.0258560180664062, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.78948, "step": 550, "tokens/total": 71933952, "tokens/train_per_sec_per_gpu": 17911.01, "tokens/trainable": 69597800 }, { "epoch": 2.779746835443038, "eval_indic_sft_mini_val_loss": 1.017600417137146, "eval_indic_sft_mini_val_runtime": 157.873, "eval_indic_sft_mini_val_samples_per_second": 12.162, "eval_indic_sft_mini_val_steps_per_second": 3.04, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 25.99, "memory/max_allocated (GiB)": 25.99, "step": 550 }, { "epoch": 2.779746835443038, "eval_tulu_sft_mini_val_loss": 2.295935869216919, "eval_tulu_sft_mini_val_runtime": 92.3573, "eval_tulu_sft_mini_val_samples_per_second": 12.473, "eval_tulu_sft_mini_val_steps_per_second": 3.118, "memory/device_reserved (GiB)": 26.59, "memory/max_active (GiB)": 25.99, "memory/max_allocated (GiB)": 25.99, "step": 550 }, { "epoch": 2.830379746835443, "grad_norm": 1.171875, "learning_rate": 1.5298029515771195e-07, "loss": 1.001215648651123, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.72159, "step": 560, "tokens/total": 73244672, "tokens/train_per_sec_per_gpu": 3958.43, "tokens/trainable": 70866632 }, { "epoch": 2.8810126582278484, "grad_norm": 1.3125, "learning_rate": 7.24046647612675e-08, "loss": 0.9905457496643066, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.6927, "step": 570, "tokens/total": 74555392, "tokens/train_per_sec_per_gpu": 17944.54, "tokens/trainable": 72132904 }, { "epoch": 2.9316455696202532, "grad_norm": 1.234375, "learning_rate": 2.156017777577346e-08, "loss": 1.0318729400634765, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.80632, "step": 580, "tokens/total": 75866112, "tokens/train_per_sec_per_gpu": 17916.79, "tokens/trainable": 73399168 }, { "epoch": 2.982278481012658, "grad_norm": 1.2421875, "learning_rate": 5.991031730367968e-10, "loss": 1.0094696044921876, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "ppl": 2.74415, "step": 590, "tokens/total": 77176832, "tokens/train_per_sec_per_gpu": 17962.72, "tokens/trainable": 74669896 }, { "epoch": 2.9873417721518987, "eval_indic_sft_mini_val_loss": 1.0181607007980347, "eval_indic_sft_mini_val_runtime": 157.7012, "eval_indic_sft_mini_val_samples_per_second": 12.175, "eval_indic_sft_mini_val_steps_per_second": 3.044, "memory/device_reserved (GiB)": 37.04, "memory/max_active (GiB)": 32.29, "memory/max_allocated (GiB)": 32.29, "step": 591 }, { "epoch": 2.9873417721518987, "eval_tulu_sft_mini_val_loss": 2.298520088195801, "eval_tulu_sft_mini_val_runtime": 92.3188, "eval_tulu_sft_mini_val_samples_per_second": 12.478, "eval_tulu_sft_mini_val_steps_per_second": 3.12, "memory/device_reserved (GiB)": 26.59, "memory/max_active (GiB)": 25.99, "memory/max_allocated (GiB)": 25.99, "step": 591 } ], "logging_steps": 10, "max_steps": 591, "num_input_tokens_seen": 0, "num_train_epochs": 3, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 1.660101173056635e+17, "train_batch_size": 4, "trial_name": null, "trial_params": null }