{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 0.026214395526076496, "eval_steps": 3000, "global_step": 3000, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.00043690659210127495, "grad_norm": 1.8826903104782104, "learning_rate": 2.4500000000000003e-05, "loss": 9.762939453125, "step": 50 }, { "epoch": 0.0008738131842025499, "grad_norm": 0.9995923042297363, "learning_rate": 4.9500000000000004e-05, "loss": 8.305597534179688, "step": 100 }, { "epoch": 0.001310719776303825, "grad_norm": 0.9061496257781982, "learning_rate": 7.45e-05, "loss": 7.173495483398438, "step": 150 }, { "epoch": 0.0017476263684050998, "grad_norm": 1.6024847030639648, "learning_rate": 9.95e-05, "loss": 6.624603271484375, "step": 200 }, { "epoch": 0.002184532960506375, "grad_norm": 1.2123334407806396, "learning_rate": 0.0001245, "loss": 6.243399658203125, "step": 250 }, { "epoch": 0.00262143955260765, "grad_norm": 1.0432322025299072, "learning_rate": 0.0001495, "loss": 5.9879296875, "step": 300 }, { "epoch": 0.0030583461447089245, "grad_norm": 0.9280146360397339, "learning_rate": 0.00017449999999999999, "loss": 5.776607666015625, "step": 350 }, { "epoch": 0.0034952527368101996, "grad_norm": 1.0289900302886963, "learning_rate": 0.00019950000000000002, "loss": 5.5920703125, "step": 400 }, { "epoch": 0.003932159328911475, "grad_norm": 1.0593407154083252, "learning_rate": 0.0002245, "loss": 5.437271728515625, "step": 450 }, { "epoch": 0.00436906592101275, "grad_norm": 1.0122662782669067, "learning_rate": 0.0002495, "loss": 5.287777709960937, "step": 500 }, { "epoch": 0.004805972513114025, "grad_norm": 0.7662743926048279, "learning_rate": 0.0002745, "loss": 5.155453491210937, "step": 550 }, { "epoch": 0.0052428791052153, "grad_norm": 0.7941539287567139, "learning_rate": 0.0002995, "loss": 5.030631713867187, "step": 600 }, { "epoch": 0.005679785697316574, "grad_norm": 0.6195730566978455, "learning_rate": 0.00032450000000000003, "loss": 4.909389953613282, "step": 650 }, { "epoch": 0.006116692289417849, "grad_norm": 0.6587674617767334, "learning_rate": 0.0003495, "loss": 4.786273498535156, "step": 700 }, { "epoch": 0.006553598881519124, "grad_norm": 0.7233771681785583, "learning_rate": 0.0003745, "loss": 4.6648342895507815, "step": 750 }, { "epoch": 0.006990505473620399, "grad_norm": 0.5489601492881775, "learning_rate": 0.0003995, "loss": 4.555048522949218, "step": 800 }, { "epoch": 0.007427412065721674, "grad_norm": 0.6154086589813232, "learning_rate": 0.0004245, "loss": 4.461051330566407, "step": 850 }, { "epoch": 0.00786431865782295, "grad_norm": 0.41333407163619995, "learning_rate": 0.00044950000000000003, "loss": 4.374936828613281, "step": 900 }, { "epoch": 0.008301225249924224, "grad_norm": 0.5036063194274902, "learning_rate": 0.0004745, "loss": 4.3033209228515625, "step": 950 }, { "epoch": 0.0087381318420255, "grad_norm": 0.3687891364097595, "learning_rate": 0.0004995, "loss": 4.233715209960938, "step": 1000 }, { "epoch": 0.009175038434126774, "grad_norm": 0.5433846116065979, "learning_rate": 0.0005245, "loss": 4.186253967285157, "step": 1050 }, { "epoch": 0.00961194502622805, "grad_norm": 0.439351350069046, "learning_rate": 0.0005495, "loss": 4.139383850097656, "step": 1100 }, { "epoch": 0.010048851618329325, "grad_norm": 0.36215880513191223, "learning_rate": 0.0005745, "loss": 4.08899658203125, "step": 1150 }, { "epoch": 0.0104857582104306, "grad_norm": 0.3970701992511749, "learning_rate": 0.0005995000000000001, "loss": 4.061238403320313, "step": 1200 }, { "epoch": 0.010922664802531873, "grad_norm": 0.3410918712615967, "learning_rate": 0.0006245000000000001, "loss": 4.026336669921875, "step": 1250 }, { "epoch": 0.011359571394633148, "grad_norm": 0.3047930598258972, "learning_rate": 0.0006495, "loss": 3.9928851318359375, "step": 1300 }, { "epoch": 0.011796477986734423, "grad_norm": 0.2819700241088867, "learning_rate": 0.0006745, "loss": 3.9584146118164063, "step": 1350 }, { "epoch": 0.012233384578835698, "grad_norm": 0.29095014929771423, "learning_rate": 0.0006995, "loss": 3.9291363525390626, "step": 1400 }, { "epoch": 0.012670291170936973, "grad_norm": 0.2987957298755646, "learning_rate": 0.0007245000000000001, "loss": 3.9090225219726564, "step": 1450 }, { "epoch": 0.013107197763038248, "grad_norm": 0.28419050574302673, "learning_rate": 0.0007495000000000001, "loss": 3.890744934082031, "step": 1500 }, { "epoch": 0.013544104355139523, "grad_norm": 0.3039053976535797, "learning_rate": 0.0007745, "loss": 3.861073303222656, "step": 1550 }, { "epoch": 0.013981010947240798, "grad_norm": 0.25676754117012024, "learning_rate": 0.0007995, "loss": 3.8428253173828124, "step": 1600 }, { "epoch": 0.014417917539342073, "grad_norm": 0.2573896646499634, "learning_rate": 0.0008245, "loss": 3.8260653686523436, "step": 1650 }, { "epoch": 0.014854824131443348, "grad_norm": 0.272127240896225, "learning_rate": 0.0008495000000000001, "loss": 3.8006341552734373, "step": 1700 }, { "epoch": 0.015291730723544623, "grad_norm": 0.23205944895744324, "learning_rate": 0.0008745000000000001, "loss": 3.7862649536132813, "step": 1750 }, { "epoch": 0.0157286373156459, "grad_norm": 0.22847425937652588, "learning_rate": 0.0008995, "loss": 3.771934814453125, "step": 1800 }, { "epoch": 0.016165543907747174, "grad_norm": 0.2651786208152771, "learning_rate": 0.0009245, "loss": 3.76009521484375, "step": 1850 }, { "epoch": 0.01660245049984845, "grad_norm": 0.24903950095176697, "learning_rate": 0.0009495, "loss": 3.7454238891601563, "step": 1900 }, { "epoch": 0.017039357091949724, "grad_norm": 0.24524764716625214, "learning_rate": 0.0009745000000000001, "loss": 3.735928955078125, "step": 1950 }, { "epoch": 0.017476263684051, "grad_norm": 0.22325558960437775, "learning_rate": 0.0009995000000000002, "loss": 3.7209417724609377, "step": 2000 }, { "epoch": 0.017913170276152274, "grad_norm": 0.23860357701778412, "learning_rate": 0.001, "loss": 3.7122769165039062, "step": 2050 }, { "epoch": 0.01835007686825355, "grad_norm": 0.2162218689918518, "learning_rate": 0.001, "loss": 3.690401611328125, "step": 2100 }, { "epoch": 0.018786983460354824, "grad_norm": 0.2171812504529953, "learning_rate": 0.001, "loss": 3.6825027465820312, "step": 2150 }, { "epoch": 0.0192238900524561, "grad_norm": 0.22677843272686005, "learning_rate": 0.001, "loss": 3.6679034423828125, "step": 2200 }, { "epoch": 0.019660796644557374, "grad_norm": 0.23934350907802582, "learning_rate": 0.001, "loss": 3.6494000244140623, "step": 2250 }, { "epoch": 0.02009770323665865, "grad_norm": 0.20819289982318878, "learning_rate": 0.001, "loss": 3.640907897949219, "step": 2300 }, { "epoch": 0.020534609828759924, "grad_norm": 0.22890810668468475, "learning_rate": 0.001, "loss": 3.6309326171875, "step": 2350 }, { "epoch": 0.0209715164208612, "grad_norm": 0.2085026353597641, "learning_rate": 0.001, "loss": 3.6261746215820314, "step": 2400 }, { "epoch": 0.021408423012962474, "grad_norm": 0.22136099636554718, "learning_rate": 0.001, "loss": 3.614878845214844, "step": 2450 }, { "epoch": 0.021845329605063746, "grad_norm": 0.21210630238056183, "learning_rate": 0.001, "loss": 3.599817810058594, "step": 2500 }, { "epoch": 0.02228223619716502, "grad_norm": 0.24638274312019348, "learning_rate": 0.001, "loss": 3.591536865234375, "step": 2550 }, { "epoch": 0.022719142789266296, "grad_norm": 0.20501184463500977, "learning_rate": 0.001, "loss": 3.58053955078125, "step": 2600 }, { "epoch": 0.02315604938136757, "grad_norm": 0.2101815640926361, "learning_rate": 0.001, "loss": 3.5824459838867186, "step": 2650 }, { "epoch": 0.023592955973468846, "grad_norm": 0.22906431555747986, "learning_rate": 0.001, "loss": 3.564821472167969, "step": 2700 }, { "epoch": 0.02402986256557012, "grad_norm": 0.20708167552947998, "learning_rate": 0.001, "loss": 3.5566305541992187, "step": 2750 }, { "epoch": 0.024466769157671396, "grad_norm": 0.21956811845302582, "learning_rate": 0.001, "loss": 3.5507748413085936, "step": 2800 }, { "epoch": 0.02490367574977267, "grad_norm": 0.24124488234519958, "learning_rate": 0.001, "loss": 3.5426095581054686, "step": 2850 }, { "epoch": 0.025340582341873946, "grad_norm": 0.20921742916107178, "learning_rate": 0.001, "loss": 3.531602783203125, "step": 2900 }, { "epoch": 0.02577748893397522, "grad_norm": 0.19601280987262726, "learning_rate": 0.001, "loss": 3.5225894165039064, "step": 2950 }, { "epoch": 0.026214395526076496, "grad_norm": 0.20243392884731293, "learning_rate": 0.001, "loss": 3.5183840942382814, "step": 3000 }, { "epoch": 0.026214395526076496, "eval_loss": 3.518667221069336, "eval_runtime": 5.8945, "eval_samples_per_second": 165.579, "eval_steps_per_second": 10.349, "step": 3000 } ], "logging_steps": 50, "max_steps": 114440, "num_input_tokens_seen": 0, "num_train_epochs": 1, "save_steps": 3000, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 3.56339612123136e+17, "train_batch_size": 16, "trial_name": null, "trial_params": null }