Files
anwgpt4-chat/checkpoint-3000/trainer_state.json
ModelHub XC 0ba3733fff 初始化项目,由ModelHub XC社区提供模型
Model: anwgpt/anwgpt4-chat
Source: Original Platform
2026-06-12 03:39:17 +08:00

455 lines
10 KiB
JSON

{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 12.0,
"eval_steps": 500,
"global_step": 3000,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.2,
"grad_norm": 2.671468734741211,
"learning_rate": 6.125e-05,
"loss": 8.0449,
"step": 50
},
{
"epoch": 0.4,
"grad_norm": 1.9384660720825195,
"learning_rate": 0.00012375,
"loss": 6.1065,
"step": 100
},
{
"epoch": 0.6,
"grad_norm": 1.7404013872146606,
"learning_rate": 0.00018625,
"loss": 5.552,
"step": 150
},
{
"epoch": 0.8,
"grad_norm": 1.5962367057800293,
"learning_rate": 0.00024875,
"loss": 5.3272,
"step": 200
},
{
"epoch": 1.0,
"grad_norm": 1.6055867671966553,
"learning_rate": 0.000245625,
"loss": 5.1701,
"step": 250
},
{
"epoch": 1.2,
"grad_norm": 1.5786716938018799,
"learning_rate": 0.0002411607142857143,
"loss": 4.7686,
"step": 300
},
{
"epoch": 1.4,
"grad_norm": 1.5948882102966309,
"learning_rate": 0.00023669642857142856,
"loss": 4.6684,
"step": 350
},
{
"epoch": 1.6,
"grad_norm": 1.5991402864456177,
"learning_rate": 0.00023223214285714286,
"loss": 4.6017,
"step": 400
},
{
"epoch": 1.8,
"grad_norm": 1.4955918788909912,
"learning_rate": 0.00022776785714285713,
"loss": 4.549,
"step": 450
},
{
"epoch": 2.0,
"grad_norm": 1.5773422718048096,
"learning_rate": 0.00022330357142857143,
"loss": 4.4926,
"step": 500
},
{
"epoch": 2.2,
"grad_norm": 1.5831668376922607,
"learning_rate": 0.00021883928571428572,
"loss": 4.1386,
"step": 550
},
{
"epoch": 2.4,
"grad_norm": 1.3561625480651855,
"learning_rate": 0.00021437500000000002,
"loss": 4.1139,
"step": 600
},
{
"epoch": 2.6,
"grad_norm": 1.4930239915847778,
"learning_rate": 0.0002099107142857143,
"loss": 4.1189,
"step": 650
},
{
"epoch": 2.8,
"grad_norm": 1.5223215818405151,
"learning_rate": 0.00020544642857142856,
"loss": 4.0464,
"step": 700
},
{
"epoch": 3.0,
"grad_norm": 1.6128135919570923,
"learning_rate": 0.00020098214285714286,
"loss": 4.0405,
"step": 750
},
{
"epoch": 3.2,
"grad_norm": 1.5877256393432617,
"learning_rate": 0.00019651785714285713,
"loss": 3.7492,
"step": 800
},
{
"epoch": 3.4,
"grad_norm": 1.5449103116989136,
"learning_rate": 0.00019205357142857143,
"loss": 3.8028,
"step": 850
},
{
"epoch": 3.6,
"grad_norm": 1.4588514566421509,
"learning_rate": 0.00018758928571428572,
"loss": 3.752,
"step": 900
},
{
"epoch": 3.8,
"grad_norm": 1.4356666803359985,
"learning_rate": 0.00018312500000000002,
"loss": 3.7031,
"step": 950
},
{
"epoch": 4.0,
"grad_norm": 1.5264209508895874,
"learning_rate": 0.0001786607142857143,
"loss": 3.7287,
"step": 1000
},
{
"epoch": 4.2,
"grad_norm": 1.351216197013855,
"learning_rate": 0.00017419642857142856,
"loss": 3.4962,
"step": 1050
},
{
"epoch": 4.4,
"grad_norm": 1.6168186664581299,
"learning_rate": 0.00016973214285714286,
"loss": 3.5006,
"step": 1100
},
{
"epoch": 4.6,
"grad_norm": 1.6266807317733765,
"learning_rate": 0.00016526785714285713,
"loss": 3.4669,
"step": 1150
},
{
"epoch": 4.8,
"grad_norm": 1.509149432182312,
"learning_rate": 0.00016080357142857142,
"loss": 3.5044,
"step": 1200
},
{
"epoch": 5.0,
"grad_norm": 1.610813856124878,
"learning_rate": 0.00015633928571428572,
"loss": 3.4626,
"step": 1250
},
{
"epoch": 5.2,
"grad_norm": 1.3284097909927368,
"learning_rate": 0.00015187500000000002,
"loss": 3.2758,
"step": 1300
},
{
"epoch": 5.4,
"grad_norm": 1.5338636636734009,
"learning_rate": 0.0001474107142857143,
"loss": 3.3102,
"step": 1350
},
{
"epoch": 5.6,
"grad_norm": 1.4522225856781006,
"learning_rate": 0.00014294642857142856,
"loss": 3.2593,
"step": 1400
},
{
"epoch": 5.8,
"grad_norm": 1.6232526302337646,
"learning_rate": 0.00013848214285714286,
"loss": 3.2785,
"step": 1450
},
{
"epoch": 6.0,
"grad_norm": 1.6043927669525146,
"learning_rate": 0.00013401785714285713,
"loss": 3.3056,
"step": 1500
},
{
"epoch": 6.2,
"grad_norm": 1.6095658540725708,
"learning_rate": 0.00012955357142857142,
"loss": 3.1076,
"step": 1550
},
{
"epoch": 6.4,
"grad_norm": 1.4529067277908325,
"learning_rate": 0.00012508928571428572,
"loss": 3.1206,
"step": 1600
},
{
"epoch": 6.6,
"grad_norm": 1.7432512044906616,
"learning_rate": 0.000120625,
"loss": 3.1245,
"step": 1650
},
{
"epoch": 6.8,
"grad_norm": 1.7623839378356934,
"learning_rate": 0.00011616071428571429,
"loss": 3.1437,
"step": 1700
},
{
"epoch": 7.0,
"grad_norm": 1.537539005279541,
"learning_rate": 0.00011169642857142857,
"loss": 3.121,
"step": 1750
},
{
"epoch": 7.2,
"grad_norm": 1.4343411922454834,
"learning_rate": 0.00010723214285714286,
"loss": 2.9752,
"step": 1800
},
{
"epoch": 7.4,
"grad_norm": 1.4659862518310547,
"learning_rate": 0.00010276785714285715,
"loss": 2.9755,
"step": 1850
},
{
"epoch": 7.6,
"grad_norm": 1.4027365446090698,
"learning_rate": 9.830357142857144e-05,
"loss": 2.9823,
"step": 1900
},
{
"epoch": 7.8,
"grad_norm": 1.6034783124923706,
"learning_rate": 9.383928571428571e-05,
"loss": 2.9971,
"step": 1950
},
{
"epoch": 8.0,
"grad_norm": 1.9349607229232788,
"learning_rate": 8.9375e-05,
"loss": 3.0408,
"step": 2000
},
{
"epoch": 8.2,
"grad_norm": 1.5146081447601318,
"learning_rate": 8.491071428571429e-05,
"loss": 2.8761,
"step": 2050
},
{
"epoch": 8.4,
"grad_norm": 1.6080121994018555,
"learning_rate": 8.044642857142857e-05,
"loss": 2.8688,
"step": 2100
},
{
"epoch": 8.6,
"grad_norm": 1.4734200239181519,
"learning_rate": 7.598214285714286e-05,
"loss": 2.8778,
"step": 2150
},
{
"epoch": 8.8,
"grad_norm": 1.5679528713226318,
"learning_rate": 7.151785714285715e-05,
"loss": 2.9088,
"step": 2200
},
{
"epoch": 9.0,
"grad_norm": 1.4920151233673096,
"learning_rate": 6.705357142857144e-05,
"loss": 2.9162,
"step": 2250
},
{
"epoch": 9.2,
"grad_norm": 1.6481536626815796,
"learning_rate": 6.25892857142857e-05,
"loss": 2.8056,
"step": 2300
},
{
"epoch": 9.4,
"grad_norm": 1.4848685264587402,
"learning_rate": 5.8125e-05,
"loss": 2.796,
"step": 2350
},
{
"epoch": 9.6,
"grad_norm": 1.376192569732666,
"learning_rate": 5.366071428571429e-05,
"loss": 2.7904,
"step": 2400
},
{
"epoch": 9.8,
"grad_norm": 1.557061791419983,
"learning_rate": 4.919642857142857e-05,
"loss": 2.8336,
"step": 2450
},
{
"epoch": 10.0,
"grad_norm": 1.642156958580017,
"learning_rate": 4.473214285714286e-05,
"loss": 2.8048,
"step": 2500
},
{
"epoch": 10.2,
"grad_norm": 1.7578301429748535,
"learning_rate": 4.026785714285714e-05,
"loss": 2.718,
"step": 2550
},
{
"epoch": 10.4,
"grad_norm": 1.5210566520690918,
"learning_rate": 3.580357142857143e-05,
"loss": 2.733,
"step": 2600
},
{
"epoch": 10.6,
"grad_norm": 1.5723469257354736,
"learning_rate": 3.133928571428572e-05,
"loss": 2.7494,
"step": 2650
},
{
"epoch": 10.8,
"grad_norm": 1.3110719919204712,
"learning_rate": 2.6875e-05,
"loss": 2.7616,
"step": 2700
},
{
"epoch": 11.0,
"grad_norm": 1.393686056137085,
"learning_rate": 2.2410714285714286e-05,
"loss": 2.7552,
"step": 2750
},
{
"epoch": 11.2,
"grad_norm": 1.5743101835250854,
"learning_rate": 1.7946428571428573e-05,
"loss": 2.6776,
"step": 2800
},
{
"epoch": 11.4,
"grad_norm": 1.7243106365203857,
"learning_rate": 1.3482142857142857e-05,
"loss": 2.7079,
"step": 2850
},
{
"epoch": 11.6,
"grad_norm": 1.5077002048492432,
"learning_rate": 9.017857142857144e-06,
"loss": 2.6959,
"step": 2900
},
{
"epoch": 11.8,
"grad_norm": 1.3862125873565674,
"learning_rate": 4.553571428571429e-06,
"loss": 2.7095,
"step": 2950
},
{
"epoch": 12.0,
"grad_norm": 1.383197546005249,
"learning_rate": 8.928571428571429e-08,
"loss": 2.7034,
"step": 3000
}
],
"logging_steps": 50,
"max_steps": 3000,
"num_input_tokens_seen": 0,
"num_train_epochs": 12,
"save_steps": 1000,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": true
},
"attributes": {}
}
},
"total_flos": 2093469401088000.0,
"train_batch_size": 2,
"trial_name": null,
"trial_params": null
}