Files
v36_v29_general_guard_lr4e7…/trainer_state.json
ModelHub XC c48b1a9c13 初始化项目,由ModelHub XC社区提供模型
Model: lldois/v36_v29_general_guard_lr4e7_ep12
Source: Original Platform
2026-07-19 04:10:09 +08:00

492 lines
12 KiB
JSON

{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 1.2000926354793886,
"eval_steps": 500,
"global_step": 648,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.018527095877721167,
"grad_norm": 9.9375,
"learning_rate": 1.8e-07,
"loss": 1.9585561752319336,
"step": 10
},
{
"epoch": 0.037054191755442334,
"grad_norm": 6.625,
"learning_rate": 3.7999999999999996e-07,
"loss": 1.8574798583984375,
"step": 20
},
{
"epoch": 0.0555812876331635,
"grad_norm": 6.09375,
"learning_rate": 3.997973287649674e-07,
"loss": 1.789466094970703,
"step": 30
},
{
"epoch": 0.07410838351088467,
"grad_norm": 10.9375,
"learning_rate": 3.9909726417418306e-07,
"loss": 1.6971963882446288,
"step": 40
},
{
"epoch": 0.09263547938860583,
"grad_norm": 5.375,
"learning_rate": 3.978990552681652e-07,
"loss": 1.9821432113647461,
"step": 50
},
{
"epoch": 0.111162575266327,
"grad_norm": 4.28125,
"learning_rate": 3.962056999834122e-07,
"loss": 1.8834493637084961,
"step": 60
},
{
"epoch": 0.12968967114404817,
"grad_norm": 6.3125,
"learning_rate": 3.9402143512002393e-07,
"loss": 1.901871681213379,
"step": 70
},
{
"epoch": 0.14821676702176934,
"grad_norm": 4.6875,
"learning_rate": 3.913517257411654e-07,
"loss": 1.947560691833496,
"step": 80
},
{
"epoch": 0.1667438628994905,
"grad_norm": 7.6875,
"learning_rate": 3.8820325149939695e-07,
"loss": 1.8572700500488282,
"step": 90
},
{
"epoch": 0.18527095877721167,
"grad_norm": 7.625,
"learning_rate": 3.8458388992408446e-07,
"loss": 1.9525575637817383,
"step": 100
},
{
"epoch": 0.20379805465493284,
"grad_norm": 7.5,
"learning_rate": 3.805026967117037e-07,
"loss": 1.9637582778930665,
"step": 110
},
{
"epoch": 0.222325150532654,
"grad_norm": 4.78125,
"learning_rate": 3.7596988306835307e-07,
"loss": 1.8187477111816406,
"step": 120
},
{
"epoch": 0.24085224641037517,
"grad_norm": 7.09375,
"learning_rate": 3.709967901611643e-07,
"loss": 1.8701536178588867,
"step": 130
},
{
"epoch": 0.25937934228809634,
"grad_norm": 5.0,
"learning_rate": 3.6559586074253323e-07,
"loss": 1.9594388961791993,
"step": 140
},
{
"epoch": 0.2779064381658175,
"grad_norm": 11.0,
"learning_rate": 3.597806080181688e-07,
"loss": 2.0100639343261717,
"step": 150
},
{
"epoch": 0.29643353404353867,
"grad_norm": 9.0,
"learning_rate": 3.535655818368514e-07,
"loss": 1.7690265655517579,
"step": 160
},
{
"epoch": 0.31496062992125984,
"grad_norm": 7.0,
"learning_rate": 3.469663322864944e-07,
"loss": 1.901205825805664,
"step": 170
},
{
"epoch": 0.333487725798981,
"grad_norm": 7.3125,
"learning_rate": 3.399993707875936e-07,
"loss": 1.8674629211425782,
"step": 180
},
{
"epoch": 0.35201482167670217,
"grad_norm": 9.375,
"learning_rate": 3.326821287814071e-07,
"loss": 1.9903881072998046,
"step": 190
},
{
"epoch": 0.37054191755442334,
"grad_norm": 13.3125,
"learning_rate": 3.250329141162305e-07,
"loss": 1.8886430740356446,
"step": 200
},
{
"epoch": 0.3890690134321445,
"grad_norm": 6.4375,
"learning_rate": 3.1707086524088837e-07,
"loss": 1.9833818435668946,
"step": 210
},
{
"epoch": 0.4075961093098657,
"grad_norm": 8.25,
"learning_rate": 3.088159033200498e-07,
"loss": 1.9257682800292968,
"step": 220
},
{
"epoch": 0.42612320518758684,
"grad_norm": 7.65625,
"learning_rate": 3.002886823911797e-07,
"loss": 1.705978012084961,
"step": 230
},
{
"epoch": 0.444650301065308,
"grad_norm": 9.75,
"learning_rate": 2.9151053768782854e-07,
"loss": 1.9222623825073242,
"step": 240
},
{
"epoch": 0.4631773969430292,
"grad_norm": 5.25,
"learning_rate": 2.8250343225856163e-07,
"loss": 1.9167638778686524,
"step": 250
},
{
"epoch": 0.48170449282075034,
"grad_norm": 14.25,
"learning_rate": 2.7328990201508475e-07,
"loss": 1.908761978149414,
"step": 260
},
{
"epoch": 0.5002315886984715,
"grad_norm": 4.71875,
"learning_rate": 2.6389299934705824e-07,
"loss": 1.913759994506836,
"step": 270
},
{
"epoch": 0.5187586845761927,
"grad_norm": 5.875,
"learning_rate": 2.5433623544467476e-07,
"loss": 1.9161806106567383,
"step": 280
},
{
"epoch": 0.5372857804539138,
"grad_norm": 4.28125,
"learning_rate": 2.446435214733126e-07,
"loss": 1.785818099975586,
"step": 290
},
{
"epoch": 0.555812876331635,
"grad_norm": 4.9375,
"learning_rate": 2.3483910874744376e-07,
"loss": 1.9147356033325196,
"step": 300
},
{
"epoch": 0.5743399722093562,
"grad_norm": 5.125,
"learning_rate": 2.2494752805348493e-07,
"loss": 1.8954235076904298,
"step": 310
},
{
"epoch": 0.5928670680870773,
"grad_norm": 8.4375,
"learning_rate": 2.14993528273405e-07,
"loss": 1.862330436706543,
"step": 320
},
{
"epoch": 0.6113941639647985,
"grad_norm": 7.15625,
"learning_rate": 2.0500201446265426e-07,
"loss": 1.8166675567626953,
"step": 330
},
{
"epoch": 0.6299212598425197,
"grad_norm": 5.78125,
"learning_rate": 1.949979855373457e-07,
"loss": 1.86541748046875,
"step": 340
},
{
"epoch": 0.6484483557202408,
"grad_norm": 5.625,
"learning_rate": 1.8500647172659497e-07,
"loss": 1.9255918502807616,
"step": 350
},
{
"epoch": 0.666975451597962,
"grad_norm": 5.71875,
"learning_rate": 1.7505247194651502e-07,
"loss": 1.8364168167114259,
"step": 360
},
{
"epoch": 0.6855025474756832,
"grad_norm": 5.53125,
"learning_rate": 1.651608912525562e-07,
"loss": 1.9211824417114258,
"step": 370
},
{
"epoch": 0.7040296433534043,
"grad_norm": 7.25,
"learning_rate": 1.553564785266874e-07,
"loss": 1.8987268447875976,
"step": 380
},
{
"epoch": 0.7225567392311255,
"grad_norm": 20.125,
"learning_rate": 1.4566376455532522e-07,
"loss": 1.8674749374389648,
"step": 390
},
{
"epoch": 0.7410838351088467,
"grad_norm": 6.25,
"learning_rate": 1.361070006529418e-07,
"loss": 1.8906829833984375,
"step": 400
},
{
"epoch": 0.7596109309865678,
"grad_norm": 11.3125,
"learning_rate": 1.267100979849152e-07,
"loss": 1.8476547241210937,
"step": 410
},
{
"epoch": 0.778138026864289,
"grad_norm": 4.6875,
"learning_rate": 1.1749656774143837e-07,
"loss": 1.9425527572631835,
"step": 420
},
{
"epoch": 0.7966651227420102,
"grad_norm": 5.15625,
"learning_rate": 1.0848946231217149e-07,
"loss": 1.7886039733886718,
"step": 430
},
{
"epoch": 0.8151922186197313,
"grad_norm": 4.65625,
"learning_rate": 9.971131760882032e-08,
"loss": 1.9002357482910157,
"step": 440
},
{
"epoch": 0.8337193144974525,
"grad_norm": 7.8125,
"learning_rate": 9.118409667995014e-08,
"loss": 2.0303293228149415,
"step": 450
},
{
"epoch": 0.8522464103751737,
"grad_norm": 6.34375,
"learning_rate": 8.292913475911168e-08,
"loss": 1.8214141845703125,
"step": 460
},
{
"epoch": 0.8707735062528948,
"grad_norm": 4.0625,
"learning_rate": 7.496708588376946e-08,
"loss": 1.7804437637329102,
"step": 470
},
{
"epoch": 0.889300602130616,
"grad_norm": 4.65625,
"learning_rate": 6.731787121859294e-08,
"loss": 1.807050895690918,
"step": 480
},
{
"epoch": 0.9078276980083372,
"grad_norm": 6.875,
"learning_rate": 6.00006292124064e-08,
"loss": 1.8698265075683593,
"step": 490
},
{
"epoch": 0.9263547938860583,
"grad_norm": 12.9375,
"learning_rate": 5.303366771350557e-08,
"loss": 1.8879680633544922,
"step": 500
},
{
"epoch": 0.9448818897637795,
"grad_norm": 7.03125,
"learning_rate": 4.6434418163148616e-08,
"loss": 1.8387239456176758,
"step": 510
},
{
"epoch": 0.9634089856415007,
"grad_norm": 4.90625,
"learning_rate": 4.021939198183113e-08,
"loss": 1.7679691314697266,
"step": 520
},
{
"epoch": 0.9819360815192218,
"grad_norm": 10.25,
"learning_rate": 3.440413925746679e-08,
"loss": 1.9303434371948243,
"step": 530
},
{
"epoch": 1.0,
"grad_norm": 10.1875,
"learning_rate": 2.900320983883575e-08,
"loss": 2.0060049057006837,
"step": 540
},
{
"epoch": 1.018527095877721,
"grad_norm": 6.3125,
"learning_rate": 2.4030116931646914e-08,
"loss": 1.8175098419189453,
"step": 550
},
{
"epoch": 1.0370541917554423,
"grad_norm": 5.53125,
"learning_rate": 1.9497303288296263e-08,
"loss": 1.8821565628051757,
"step": 560
},
{
"epoch": 1.0555812876331636,
"grad_norm": 7.5,
"learning_rate": 1.54161100759155e-08,
"loss": 1.9253089904785157,
"step": 570
},
{
"epoch": 1.0741083835108847,
"grad_norm": 6.9375,
"learning_rate": 1.1796748500603015e-08,
"loss": 1.918222999572754,
"step": 580
},
{
"epoch": 1.0926354793886057,
"grad_norm": 5.71875,
"learning_rate": 8.648274258834565e-09,
"loss": 1.950904655456543,
"step": 590
},
{
"epoch": 1.111162575266327,
"grad_norm": 11.25,
"learning_rate": 5.978564879976078e-09,
"loss": 1.9175649642944337,
"step": 600
},
{
"epoch": 1.1296896711440483,
"grad_norm": 12.6875,
"learning_rate": 3.79430001658787e-09,
"loss": 1.9034099578857422,
"step": 610
},
{
"epoch": 1.1482167670217693,
"grad_norm": 5.0625,
"learning_rate": 2.100944731834797e-09,
"loss": 1.9379833221435547,
"step": 620
},
{
"epoch": 1.1667438628994904,
"grad_norm": 5.8125,
"learning_rate": 9.027358258169026e-10,
"loss": 1.955449676513672,
"step": 630
},
{
"epoch": 1.1852709587772117,
"grad_norm": 4.78125,
"learning_rate": 2.0267123503259208e-10,
"loss": 1.7997220993041991,
"step": 640
},
{
"epoch": 1.2000926354793886,
"step": 648,
"total_flos": 3.10173216276566e+17,
"train_loss": 1.8869445338661288,
"train_runtime": 5074.5206,
"train_samples_per_second": 0.511,
"train_steps_per_second": 0.128
}
],
"logging_steps": 10,
"max_steps": 648,
"num_input_tokens_seen": 0,
"num_train_epochs": 2,
"save_steps": 500,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": false,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 3.10173216276566e+17,
"train_batch_size": 1,
"trial_name": null,
"trial_params": null
}