{ "best_metric": null, "best_model_checkpoint": null, "epoch": 0.9996634885025238, "eval_steps": 500, "global_step": 1114, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.0089736399326977, "grad_norm": 2.1750009059906006, "learning_rate": 2e-05, "loss": 1.9421, "step": 10 }, { "epoch": 0.0179472798653954, "grad_norm": 1.2129733562469482, "learning_rate": 4e-05, "loss": 1.6421, "step": 20 }, { "epoch": 0.0269209197980931, "grad_norm": 1.2686073780059814, "learning_rate": 6e-05, "loss": 1.6084, "step": 30 }, { "epoch": 0.0358945597307908, "grad_norm": 0.8401476740837097, "learning_rate": 8e-05, "loss": 1.4811, "step": 40 }, { "epoch": 0.044868199663488505, "grad_norm": 1.3034749031066895, "learning_rate": 0.0001, "loss": 1.4327, "step": 50 }, { "epoch": 0.0538418395961862, "grad_norm": 0.6824946403503418, "learning_rate": 0.00012, "loss": 1.4694, "step": 60 }, { "epoch": 0.0628154795288839, "grad_norm": 0.670620858669281, "learning_rate": 0.00014, "loss": 1.3643, "step": 70 }, { "epoch": 0.0717891194615816, "grad_norm": 0.8601792454719543, "learning_rate": 0.00016, "loss": 1.3669, "step": 80 }, { "epoch": 0.08076275939427931, "grad_norm": 0.6021562814712524, "learning_rate": 0.00018, "loss": 1.3697, "step": 90 }, { "epoch": 0.08973639932697701, "grad_norm": 0.672222912311554, "learning_rate": 0.0002, "loss": 1.3505, "step": 100 }, { "epoch": 0.09871003925967471, "grad_norm": 0.5645850896835327, "learning_rate": 0.00019995200907733468, "loss": 1.3051, "step": 110 }, { "epoch": 0.1076836791923724, "grad_norm": 0.6567140817642212, "learning_rate": 0.00019980808237191178, "loss": 1.3072, "step": 120 }, { "epoch": 0.1166573191250701, "grad_norm": 0.5479955077171326, "learning_rate": 0.00019956835802723916, "loss": 1.42, "step": 130 }, { "epoch": 0.1256309590577678, "grad_norm": 0.710563600063324, "learning_rate": 0.0001992330661351665, "loss": 1.318, "step": 140 }, { "epoch": 0.1346045989904655, "grad_norm": 0.5019824504852295, "learning_rate": 0.00019880252851503915, "loss": 1.3518, "step": 150 }, { "epoch": 0.1435782389231632, "grad_norm": 0.783736526966095, "learning_rate": 0.0001982771584048096, "loss": 1.3841, "step": 160 }, { "epoch": 0.1525518788558609, "grad_norm": 0.5478121042251587, "learning_rate": 0.00019765746006440455, "loss": 1.2956, "step": 170 }, { "epoch": 0.16152551878855861, "grad_norm": 0.675541877746582, "learning_rate": 0.00019694402829172663, "loss": 1.4014, "step": 180 }, { "epoch": 0.17049915872125632, "grad_norm": 0.7109649777412415, "learning_rate": 0.0001961375478517564, "loss": 1.2814, "step": 190 }, { "epoch": 0.17947279865395402, "grad_norm": 0.6496264338493347, "learning_rate": 0.00019523879281930235, "loss": 1.33, "step": 200 }, { "epoch": 0.18844643858665172, "grad_norm": 0.7981849908828735, "learning_rate": 0.00019424862583602965, "loss": 1.3878, "step": 210 }, { "epoch": 0.19742007851934942, "grad_norm": 0.5120641589164734, "learning_rate": 0.00019316799728248075, "loss": 1.2728, "step": 220 }, { "epoch": 0.2063937184520471, "grad_norm": 0.5666424632072449, "learning_rate": 0.00019199794436588243, "loss": 1.3446, "step": 230 }, { "epoch": 0.2153673583847448, "grad_norm": 0.6776772141456604, "learning_rate": 0.00019073959012461545, "loss": 1.3373, "step": 240 }, { "epoch": 0.2243409983174425, "grad_norm": 0.7518035173416138, "learning_rate": 0.00018939414235030134, "loss": 1.3381, "step": 250 }, { "epoch": 0.2333146382501402, "grad_norm": 0.6173065304756165, "learning_rate": 0.0001879628924285419, "loss": 1.3331, "step": 260 }, { "epoch": 0.2422882781828379, "grad_norm": 0.6449307203292847, "learning_rate": 0.00018644721409942323, "loss": 1.3048, "step": 270 }, { "epoch": 0.2512619181155356, "grad_norm": 0.7187690734863281, "learning_rate": 0.00018484856213897498, "loss": 1.3328, "step": 280 }, { "epoch": 0.26023555804823334, "grad_norm": 0.5725300312042236, "learning_rate": 0.00018316847096284917, "loss": 1.4003, "step": 290 }, { "epoch": 0.269209197980931, "grad_norm": 5.041274070739746, "learning_rate": 0.0001814085531535599, "loss": 1.3066, "step": 300 }, { "epoch": 0.2781828379136287, "grad_norm": 0.4397372603416443, "learning_rate": 0.00017957049791269685, "loss": 1.3193, "step": 310 }, { "epoch": 0.2871564778463264, "grad_norm": 0.5465174913406372, "learning_rate": 0.00017765606943959833, "loss": 1.2508, "step": 320 }, { "epoch": 0.2961301177790241, "grad_norm": 0.5439965724945068, "learning_rate": 0.00017566710523804043, "loss": 1.3258, "step": 330 }, { "epoch": 0.3051037577117218, "grad_norm": 0.6401047110557556, "learning_rate": 0.00017360551435256674, "loss": 1.3563, "step": 340 }, { "epoch": 0.3140773976444195, "grad_norm": 0.5497185587882996, "learning_rate": 0.0001714732755361523, "loss": 1.267, "step": 350 }, { "epoch": 0.32305103757711723, "grad_norm": 0.6040382981300354, "learning_rate": 0.00016927243535095997, "loss": 1.3057, "step": 360 }, { "epoch": 0.3320246775098149, "grad_norm": 0.6971787810325623, "learning_rate": 0.00016700510620401224, "loss": 1.4139, "step": 370 }, { "epoch": 0.34099831744251263, "grad_norm": 0.5650486946105957, "learning_rate": 0.00016467346431966413, "loss": 1.3388, "step": 380 }, { "epoch": 0.3499719573752103, "grad_norm": 0.5824179649353027, "learning_rate": 0.00016227974765082274, "loss": 1.3194, "step": 390 }, { "epoch": 0.35894559730790804, "grad_norm": 0.4857507646083832, "learning_rate": 0.00015982625373091875, "loss": 1.3274, "step": 400 }, { "epoch": 0.3679192372406057, "grad_norm": 0.49048104882240295, "learning_rate": 0.00015731533746869163, "loss": 1.2175, "step": 410 }, { "epoch": 0.37689287717330344, "grad_norm": 0.5631568431854248, "learning_rate": 0.00015474940888790455, "loss": 1.2392, "step": 420 }, { "epoch": 0.3858665171060011, "grad_norm": 0.5326385498046875, "learning_rate": 0.0001521309308141592, "loss": 1.3241, "step": 430 }, { "epoch": 0.39484015703869885, "grad_norm": 0.4978722631931305, "learning_rate": 0.00014946241651103034, "loss": 1.2032, "step": 440 }, { "epoch": 0.4038137969713965, "grad_norm": 0.4760749936103821, "learning_rate": 0.00014674642726778906, "loss": 1.2336, "step": 450 }, { "epoch": 0.4127874369040942, "grad_norm": 0.5480512380599976, "learning_rate": 0.00014398556994102996, "loss": 1.3644, "step": 460 }, { "epoch": 0.4217610768367919, "grad_norm": 0.5921517014503479, "learning_rate": 0.00014118249445256223, "loss": 1.2761, "step": 470 }, { "epoch": 0.4307347167694896, "grad_norm": 0.505266547203064, "learning_rate": 0.00013833989124596572, "loss": 1.2971, "step": 480 }, { "epoch": 0.43970835670218733, "grad_norm": 0.432982861995697, "learning_rate": 0.00013546048870425356, "loss": 1.2217, "step": 490 }, { "epoch": 0.448681996634885, "grad_norm": 0.5738311409950256, "learning_rate": 0.0001325470505311198, "loss": 1.2898, "step": 500 }, { "epoch": 0.45765563656758274, "grad_norm": 0.6067373752593994, "learning_rate": 0.0001296023730982855, "loss": 1.3245, "step": 510 }, { "epoch": 0.4666292765002804, "grad_norm": 0.5424025058746338, "learning_rate": 0.00012662928276148985, "loss": 1.2277, "step": 520 }, { "epoch": 0.47560291643297814, "grad_norm": 0.5827418565750122, "learning_rate": 0.00012363063314770135, "loss": 1.2828, "step": 530 }, { "epoch": 0.4845765563656758, "grad_norm": 0.5253446102142334, "learning_rate": 0.0001206093024161544, "loss": 1.2969, "step": 540 }, { "epoch": 0.49355019629837354, "grad_norm": 0.6067509055137634, "learning_rate": 0.00011756819049583861, "loss": 1.3125, "step": 550 }, { "epoch": 0.5025238362310712, "grad_norm": 0.5548892617225647, "learning_rate": 0.00011451021630209371, "loss": 1.2726, "step": 560 }, { "epoch": 0.511497476163769, "grad_norm": 0.6270582675933838, "learning_rate": 0.0001114383149349806, "loss": 1.241, "step": 570 }, { "epoch": 0.5204711160964667, "grad_norm": 0.5589553117752075, "learning_rate": 0.00010835543486211815, "loss": 1.2897, "step": 580 }, { "epoch": 0.5294447560291643, "grad_norm": 0.5768513083457947, "learning_rate": 0.00010526453508868961, "loss": 1.266, "step": 590 }, { "epoch": 0.538418395961862, "grad_norm": 0.6176390051841736, "learning_rate": 0.00010216858231733488, "loss": 1.2771, "step": 600 }, { "epoch": 0.5473920358945598, "grad_norm": 0.5259169340133667, "learning_rate": 9.907054810065446e-05, "loss": 1.314, "step": 610 }, { "epoch": 0.5563656758272574, "grad_norm": 0.48984938859939575, "learning_rate": 9.597340598905852e-05, "loss": 1.2227, "step": 620 }, { "epoch": 0.5653393157599551, "grad_norm": 0.504762589931488, "learning_rate": 9.28801286766982e-05, "loss": 1.2152, "step": 630 }, { "epoch": 0.5743129556926528, "grad_norm": 0.6546471118927002, "learning_rate": 8.979368514821916e-05, "loss": 1.2996, "step": 640 }, { "epoch": 0.5832865956253506, "grad_norm": 0.6191439032554626, "learning_rate": 8.671703782907518e-05, "loss": 1.2966, "step": 650 }, { "epoch": 0.5922602355580482, "grad_norm": 0.4806272089481354, "learning_rate": 8.365313974213737e-05, "loss": 1.2665, "step": 660 }, { "epoch": 0.6012338754907459, "grad_norm": 0.4365353286266327, "learning_rate": 8.060493167332874e-05, "loss": 1.2466, "step": 670 }, { "epoch": 0.6102075154234436, "grad_norm": 0.5589653849601746, "learning_rate": 7.757533934900316e-05, "loss": 1.2992, "step": 680 }, { "epoch": 0.6191811553561414, "grad_norm": 0.6361960172653198, "learning_rate": 7.456727062777958e-05, "loss": 1.2096, "step": 690 }, { "epoch": 0.628154795288839, "grad_norm": 0.6086412072181702, "learning_rate": 7.15836127095254e-05, "loss": 1.3301, "step": 700 }, { "epoch": 0.6371284352215367, "grad_norm": 0.5902493596076965, "learning_rate": 6.862722936416897e-05, "loss": 1.3265, "step": 710 }, { "epoch": 0.6461020751542345, "grad_norm": 0.4792775809764862, "learning_rate": 6.570095818300012e-05, "loss": 1.2416, "step": 720 }, { "epoch": 0.6550757150869322, "grad_norm": 0.48689934611320496, "learning_rate": 6.280760785509801e-05, "loss": 1.2668, "step": 730 }, { "epoch": 0.6640493550196298, "grad_norm": 0.5720877051353455, "learning_rate": 5.9949955471499186e-05, "loss": 1.2979, "step": 740 }, { "epoch": 0.6730229949523275, "grad_norm": 0.6165252327919006, "learning_rate": 5.713074385969457e-05, "loss": 1.2269, "step": 750 }, { "epoch": 0.6819966348850253, "grad_norm": 0.554986298084259, "learning_rate": 5.435267895101302e-05, "loss": 1.3656, "step": 760 }, { "epoch": 0.6909702748177229, "grad_norm": 0.6707109808921814, "learning_rate": 5.161842718341825e-05, "loss": 1.3012, "step": 770 }, { "epoch": 0.6999439147504206, "grad_norm": 0.45011788606643677, "learning_rate": 4.8930612942212916e-05, "loss": 1.2375, "step": 780 }, { "epoch": 0.7089175546831183, "grad_norm": 0.5251748561859131, "learning_rate": 4.629181604110464e-05, "loss": 1.2007, "step": 790 }, { "epoch": 0.7178911946158161, "grad_norm": 0.5204996466636658, "learning_rate": 4.3704569246053805e-05, "loss": 1.3049, "step": 800 }, { "epoch": 0.7268648345485137, "grad_norm": 0.6533709764480591, "learning_rate": 4.1171355844277394e-05, "loss": 1.293, "step": 810 }, { "epoch": 0.7358384744812114, "grad_norm": 0.589524507522583, "learning_rate": 3.869460726074474e-05, "loss": 1.3565, "step": 820 }, { "epoch": 0.7448121144139092, "grad_norm": 0.6437689661979675, "learning_rate": 3.6276700724450384e-05, "loss": 1.2496, "step": 830 }, { "epoch": 0.7537857543466069, "grad_norm": 0.634338915348053, "learning_rate": 3.391995698670638e-05, "loss": 1.2889, "step": 840 }, { "epoch": 0.7627593942793045, "grad_norm": 0.5487192869186401, "learning_rate": 3.162663809364178e-05, "loss": 1.204, "step": 850 }, { "epoch": 0.7717330342120022, "grad_norm": 0.5915932655334473, "learning_rate": 2.9398945215049567e-05, "loss": 1.2491, "step": 860 }, { "epoch": 0.7807066741447, "grad_norm": 0.5967379212379456, "learning_rate": 2.7239016531662887e-05, "loss": 1.2519, "step": 870 }, { "epoch": 0.7896803140773977, "grad_norm": 0.589465320110321, "learning_rate": 2.514892518288988e-05, "loss": 1.3233, "step": 880 }, { "epoch": 0.7986539540100953, "grad_norm": 0.625720202922821, "learning_rate": 2.3130677276976232e-05, "loss": 1.2831, "step": 890 }, { "epoch": 0.807627593942793, "grad_norm": 0.5853390693664551, "learning_rate": 2.118620996550529e-05, "loss": 1.2793, "step": 900 }, { "epoch": 0.8166012338754908, "grad_norm": 0.5645495653152466, "learning_rate": 1.9317389584084568e-05, "loss": 1.2646, "step": 910 }, { "epoch": 0.8255748738081884, "grad_norm": 0.5386382341384888, "learning_rate": 1.7526009861001956e-05, "loss": 1.2479, "step": 920 }, { "epoch": 0.8345485137408861, "grad_norm": 0.5386704802513123, "learning_rate": 1.5813790195572674e-05, "loss": 1.2772, "step": 930 }, { "epoch": 0.8435221536735839, "grad_norm": 0.6007257103919983, "learning_rate": 1.4182374007827603e-05, "loss": 1.2606, "step": 940 }, { "epoch": 0.8524957936062816, "grad_norm": 0.6834496259689331, "learning_rate": 1.263332716112885e-05, "loss": 1.2778, "step": 950 }, { "epoch": 0.8614694335389792, "grad_norm": 0.5198489427566528, "learning_rate": 1.1168136459224842e-05, "loss": 1.2838, "step": 960 }, { "epoch": 0.8704430734716769, "grad_norm": 0.5519130825996399, "learning_rate": 9.788208219188932e-06, "loss": 1.2267, "step": 970 }, { "epoch": 0.8794167134043747, "grad_norm": 0.5107753276824951, "learning_rate": 8.494866921610133e-06, "loss": 1.2613, "step": 980 }, { "epoch": 0.8883903533370724, "grad_norm": 0.4735373556613922, "learning_rate": 7.289353939332288e-06, "loss": 1.2303, "step": 990 }, { "epoch": 0.89736399326977, "grad_norm": 0.4740982949733734, "learning_rate": 6.1728263459614796e-06, "loss": 1.2695, "step": 1000 }, { "epoch": 0.9063376332024677, "grad_norm": 0.580386221408844, "learning_rate": 5.146355805285452e-06, "loss": 1.2746, "step": 1010 }, { "epoch": 0.9153112731351655, "grad_norm": 0.5270484089851379, "learning_rate": 4.210927542670917e-06, "loss": 1.3423, "step": 1020 }, { "epoch": 0.9242849130678632, "grad_norm": 0.6918372511863708, "learning_rate": 3.367439399426087e-06, "loss": 1.2658, "step": 1030 }, { "epoch": 0.9332585530005608, "grad_norm": 0.5074269771575928, "learning_rate": 2.616700971036001e-06, "loss": 1.2453, "step": 1040 }, { "epoch": 0.9422321929332585, "grad_norm": 0.5192223191261292, "learning_rate": 1.959432830097807e-06, "loss": 1.2882, "step": 1050 }, { "epoch": 0.9512058328659563, "grad_norm": 0.6576820611953735, "learning_rate": 1.396265834701982e-06, "loss": 1.3017, "step": 1060 }, { "epoch": 0.9601794727986539, "grad_norm": 0.6131830215454102, "learning_rate": 9.277405229229708e-07, "loss": 1.2788, "step": 1070 }, { "epoch": 0.9691531127313516, "grad_norm": 0.7722458839416504, "learning_rate": 5.543065940008862e-07, "loss": 1.3333, "step": 1080 }, { "epoch": 0.9781267526640494, "grad_norm": 0.5115715265274048, "learning_rate": 2.7632247671177667e-07, "loss": 1.2451, "step": 1090 }, { "epoch": 0.9871003925967471, "grad_norm": 0.4745628237724304, "learning_rate": 9.405498534115209e-08, "loss": 1.2909, "step": 1100 }, { "epoch": 0.9960740325294447, "grad_norm": 0.5007497668266296, "learning_rate": 7.679063590670942e-09, "loss": 1.274, "step": 1110 } ], "logging_steps": 10, "max_steps": 1114, "num_input_tokens_seen": 0, "num_train_epochs": 1, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 2.558882397416325e+17, "train_batch_size": 1, "trial_name": null, "trial_params": null }