{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 2.0, "eval_steps": 500, "global_step": 1374, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.07285974499089254, "grad_norm": 0.33655378222465515, "learning_rate": 0.00019428152492668624, "loss": 1.248601837158203, "step": 50 }, { "epoch": 0.14571948998178508, "grad_norm": 0.3381595313549042, "learning_rate": 0.00018695014662756598, "loss": 0.8999367523193359, "step": 100 }, { "epoch": 0.2185792349726776, "grad_norm": 0.30710718035697937, "learning_rate": 0.00017961876832844575, "loss": 0.8563268280029297, "step": 150 }, { "epoch": 0.29143897996357016, "grad_norm": 0.39056596159935, "learning_rate": 0.0001722873900293255, "loss": 0.8502767181396484, "step": 200 }, { "epoch": 0.36429872495446264, "grad_norm": 0.35749396681785583, "learning_rate": 0.0001649560117302053, "loss": 0.8379911804199218, "step": 250 }, { "epoch": 0.4371584699453552, "grad_norm": 0.41779738664627075, "learning_rate": 0.00015762463343108504, "loss": 0.8131264495849609, "step": 300 }, { "epoch": 0.5100182149362478, "grad_norm": 0.40256431698799133, "learning_rate": 0.0001502932551319648, "loss": 0.8029293060302735, "step": 350 }, { "epoch": 0.5828779599271403, "grad_norm": 0.39221078157424927, "learning_rate": 0.00014296187683284457, "loss": 0.7728028106689453, "step": 400 }, { "epoch": 0.6557377049180327, "grad_norm": 0.3179788887500763, "learning_rate": 0.00013563049853372434, "loss": 0.775037841796875, "step": 450 }, { "epoch": 0.7285974499089253, "grad_norm": 0.4592280685901642, "learning_rate": 0.0001282991202346041, "loss": 0.7742440032958985, "step": 500 }, { "epoch": 0.8014571948998178, "grad_norm": 0.46030479669570923, "learning_rate": 0.00012096774193548388, "loss": 0.7653812408447266, "step": 550 }, { "epoch": 0.8743169398907104, "grad_norm": 0.47871556878089905, "learning_rate": 0.00011363636363636365, "loss": 0.7765991973876953, "step": 600 }, { "epoch": 0.9471766848816029, "grad_norm": 0.4337916076183319, "learning_rate": 0.0001063049853372434, "loss": 0.7484037017822266, "step": 650 }, { "epoch": 1.018943533697632, "grad_norm": 0.34845250844955444, "learning_rate": 9.897360703812317e-05, "loss": 0.710738525390625, "step": 700 }, { "epoch": 1.0918032786885246, "grad_norm": 0.5349538922309875, "learning_rate": 9.164222873900293e-05, "loss": 0.6187635040283204, "step": 750 }, { "epoch": 1.164663023679417, "grad_norm": 0.4864880442619324, "learning_rate": 8.431085043988271e-05, "loss": 0.640613784790039, "step": 800 }, { "epoch": 1.2375227686703096, "grad_norm": 0.5053865909576416, "learning_rate": 7.697947214076246e-05, "loss": 0.6253266906738282, "step": 850 }, { "epoch": 1.3103825136612022, "grad_norm": 0.5343588590621948, "learning_rate": 6.964809384164224e-05, "loss": 0.6396315383911133, "step": 900 }, { "epoch": 1.3832422586520947, "grad_norm": 0.46341976523399353, "learning_rate": 6.2316715542522e-05, "loss": 0.6397412490844726, "step": 950 }, { "epoch": 1.4561020036429873, "grad_norm": 0.6249243021011353, "learning_rate": 5.498533724340176e-05, "loss": 0.6349585723876953, "step": 1000 }, { "epoch": 1.5289617486338798, "grad_norm": 0.5683321356773376, "learning_rate": 4.765395894428153e-05, "loss": 0.6123217391967773, "step": 1050 }, { "epoch": 1.6018214936247723, "grad_norm": 0.5973424911499023, "learning_rate": 4.032258064516129e-05, "loss": 0.6341474533081055, "step": 1100 }, { "epoch": 1.6746812386156649, "grad_norm": 0.540243923664093, "learning_rate": 3.2991202346041056e-05, "loss": 0.6104950714111328, "step": 1150 }, { "epoch": 1.7475409836065574, "grad_norm": 0.47242051362991333, "learning_rate": 2.565982404692082e-05, "loss": 0.5970580291748047, "step": 1200 }, { "epoch": 1.82040072859745, "grad_norm": 0.651258111000061, "learning_rate": 1.8328445747800586e-05, "loss": 0.629194221496582, "step": 1250 }, { "epoch": 1.8932604735883425, "grad_norm": 0.6018247604370117, "learning_rate": 1.0997067448680352e-05, "loss": 0.6117166137695312, "step": 1300 }, { "epoch": 1.966120218579235, "grad_norm": 0.541965901851654, "learning_rate": 3.6656891495601177e-06, "loss": 0.6098161315917969, "step": 1350 } ], "logging_steps": 50, "max_steps": 1374, "num_input_tokens_seen": 0, "num_train_epochs": 2, "save_steps": 200, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 8.08676495418409e+16, "train_batch_size": 2, "trial_name": null, "trial_params": null }