初始化项目,由ModelHub XC社区提供模型

Model: Henry236/nilechat-eg-stage1-general
Source: Original Platform
This commit is contained in:
ModelHub XC
2026-08-29 23:44:17 +08:00
commit c3daf6ac60
23 changed files with 53047 additions and 0 deletions

View File

@@ -0,0 +1,426 @@
{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 1.0,
"eval_steps": 500,
"global_step": 563,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.017777777777777778,
"grad_norm": 45.25,
"learning_rate": 1.5789473684210526e-06,
"loss": 3.1581,
"step": 10
},
{
"epoch": 0.035555555555555556,
"grad_norm": 33.25,
"learning_rate": 3.3333333333333333e-06,
"loss": 3.0099,
"step": 20
},
{
"epoch": 0.05333333333333334,
"grad_norm": 29.25,
"learning_rate": 5.087719298245615e-06,
"loss": 2.8287,
"step": 30
},
{
"epoch": 0.07111111111111111,
"grad_norm": 21.25,
"learning_rate": 6.842105263157896e-06,
"loss": 2.588,
"step": 40
},
{
"epoch": 0.08888888888888889,
"grad_norm": 22.75,
"learning_rate": 8.596491228070176e-06,
"loss": 2.2516,
"step": 50
},
{
"epoch": 0.10666666666666667,
"grad_norm": 21.875,
"learning_rate": 9.999614527738882e-06,
"loss": 1.9088,
"step": 60
},
{
"epoch": 0.12444444444444444,
"grad_norm": 16.5,
"learning_rate": 9.986129238305635e-06,
"loss": 1.7622,
"step": 70
},
{
"epoch": 0.14222222222222222,
"grad_norm": 14.9375,
"learning_rate": 9.953429730181653e-06,
"loss": 1.6773,
"step": 80
},
{
"epoch": 0.16,
"grad_norm": 17.75,
"learning_rate": 9.901642012034214e-06,
"loss": 1.6001,
"step": 90
},
{
"epoch": 0.17777777777777778,
"grad_norm": 13.8125,
"learning_rate": 9.830965649597455e-06,
"loss": 1.6067,
"step": 100
},
{
"epoch": 0.19555555555555557,
"grad_norm": 17.25,
"learning_rate": 9.741672996639046e-06,
"loss": 1.6422,
"step": 110
},
{
"epoch": 0.21333333333333335,
"grad_norm": 17.5,
"learning_rate": 9.634108145435665e-06,
"loss": 1.615,
"step": 120
},
{
"epoch": 0.2311111111111111,
"grad_norm": 12.9375,
"learning_rate": 9.508685600801704e-06,
"loss": 1.5367,
"step": 130
},
{
"epoch": 0.24888888888888888,
"grad_norm": 12.0,
"learning_rate": 9.365888682780862e-06,
"loss": 1.564,
"step": 140
},
{
"epoch": 0.26666666666666666,
"grad_norm": 13.8125,
"learning_rate": 9.206267664155906e-06,
"loss": 1.5768,
"step": 150
},
{
"epoch": 0.28444444444444444,
"grad_norm": 11.4375,
"learning_rate": 9.03043764995379e-06,
"loss": 1.5899,
"step": 160
},
{
"epoch": 0.3022222222222222,
"grad_norm": 10.5,
"learning_rate": 8.839076207117485e-06,
"loss": 1.56,
"step": 170
},
{
"epoch": 0.32,
"grad_norm": 9.875,
"learning_rate": 8.63292075347872e-06,
"loss": 1.5181,
"step": 180
},
{
"epoch": 0.3377777777777778,
"grad_norm": 15.0625,
"learning_rate": 8.412765716093273e-06,
"loss": 1.5574,
"step": 190
},
{
"epoch": 0.35555555555555557,
"grad_norm": 11.75,
"learning_rate": 8.179459469889269e-06,
"loss": 1.5206,
"step": 200
},
{
"epoch": 0.37333333333333335,
"grad_norm": 19.0,
"learning_rate": 7.933901068425539e-06,
"loss": 1.4926,
"step": 210
},
{
"epoch": 0.39111111111111113,
"grad_norm": 10.625,
"learning_rate": 7.67703677935813e-06,
"loss": 1.5695,
"step": 220
},
{
"epoch": 0.4088888888888889,
"grad_norm": 14.375,
"learning_rate": 7.40985643796569e-06,
"loss": 1.5674,
"step": 230
},
{
"epoch": 0.4266666666666667,
"grad_norm": 10.0625,
"learning_rate": 7.133389632785543e-06,
"loss": 1.5065,
"step": 240
},
{
"epoch": 0.4444444444444444,
"grad_norm": 9.125,
"learning_rate": 6.8487017380592266e-06,
"loss": 1.4844,
"step": 250
},
{
"epoch": 0.4622222222222222,
"grad_norm": 10.375,
"learning_rate": 6.5568898082765945e-06,
"loss": 1.47,
"step": 260
},
{
"epoch": 0.48,
"grad_norm": 12.3125,
"learning_rate": 6.25907835063901e-06,
"loss": 1.4825,
"step": 270
},
{
"epoch": 0.49777777777777776,
"grad_norm": 10.5,
"learning_rate": 5.9564149917325845e-06,
"loss": 1.5384,
"step": 280
},
{
"epoch": 0.5155555555555555,
"grad_norm": 10.875,
"learning_rate": 5.650066055110067e-06,
"loss": 1.5228,
"step": 290
},
{
"epoch": 0.5333333333333333,
"grad_norm": 9.5625,
"learning_rate": 5.341212066823356e-06,
"loss": 1.5414,
"step": 300
},
{
"epoch": 0.5511111111111111,
"grad_norm": 10.5,
"learning_rate": 5.0310432062261764e-06,
"loss": 1.4797,
"step": 310
},
{
"epoch": 0.5688888888888889,
"grad_norm": 16.125,
"learning_rate": 4.720754719577448e-06,
"loss": 1.5175,
"step": 320
},
{
"epoch": 0.5866666666666667,
"grad_norm": 10.9375,
"learning_rate": 4.41154231411915e-06,
"loss": 1.512,
"step": 330
},
{
"epoch": 0.6044444444444445,
"grad_norm": 8.4375,
"learning_rate": 4.104597550377776e-06,
"loss": 1.4381,
"step": 340
},
{
"epoch": 0.6222222222222222,
"grad_norm": 13.4375,
"learning_rate": 3.8011032504453e-06,
"loss": 1.4815,
"step": 350
},
{
"epoch": 0.64,
"grad_norm": 10.9375,
"learning_rate": 3.5022289399339933e-06,
"loss": 1.4902,
"step": 360
},
{
"epoch": 0.6577777777777778,
"grad_norm": 12.25,
"learning_rate": 3.209126341169681e-06,
"loss": 1.4877,
"step": 370
},
{
"epoch": 0.6755555555555556,
"grad_norm": 11.25,
"learning_rate": 2.9229249349905686e-06,
"loss": 1.5034,
"step": 380
},
{
"epoch": 0.6933333333333334,
"grad_norm": 10.6875,
"learning_rate": 2.644727608254396e-06,
"loss": 1.4798,
"step": 390
},
{
"epoch": 0.7111111111111111,
"grad_norm": 11.125,
"learning_rate": 2.3756064038264033e-06,
"loss": 1.4458,
"step": 400
},
{
"epoch": 0.7288888888888889,
"grad_norm": 10.5625,
"learning_rate": 2.1165983894256647e-06,
"loss": 1.3626,
"step": 410
},
{
"epoch": 0.7466666666666667,
"grad_norm": 9.3125,
"learning_rate": 1.8687016612493542e-06,
"loss": 1.4922,
"step": 420
},
{
"epoch": 0.7644444444444445,
"grad_norm": 10.5625,
"learning_rate": 1.6328714977750698e-06,
"loss": 1.4503,
"step": 430
},
{
"epoch": 0.7822222222222223,
"grad_norm": 10.625,
"learning_rate": 1.4100166785627301e-06,
"loss": 1.4836,
"step": 440
},
{
"epoch": 0.8,
"grad_norm": 12.0625,
"learning_rate": 1.2009959822416012e-06,
"loss": 1.4874,
"step": 450
},
{
"epoch": 0.8177777777777778,
"grad_norm": 9.4375,
"learning_rate": 1.006614877177638e-06,
"loss": 1.4905,
"step": 460
},
{
"epoch": 0.8355555555555556,
"grad_norm": 10.25,
"learning_rate": 8.276224175737152e-07,
"loss": 1.5102,
"step": 470
},
{
"epoch": 0.8533333333333334,
"grad_norm": 10.5625,
"learning_rate": 6.647083569637797e-07,
"loss": 1.5007,
"step": 480
},
{
"epoch": 0.8711111111111111,
"grad_norm": 14.0,
"learning_rate": 5.185004902241241e-07,
"loss": 1.4803,
"step": 490
},
{
"epoch": 0.8888888888888888,
"grad_norm": 10.25,
"learning_rate": 3.8956223434447936e-07,
"loss": 1.4122,
"step": 500
},
{
"epoch": 0.9066666666666666,
"grad_norm": 9.8125,
"learning_rate": 2.783904572814622e-07,
"loss": 1.4984,
"step": 510
},
{
"epoch": 0.9244444444444444,
"grad_norm": 11.5625,
"learning_rate": 1.8541356326100436e-07,
"loss": 1.5673,
"step": 520
},
{
"epoch": 0.9422222222222222,
"grad_norm": 10.8125,
"learning_rate": 1.1098984190808403e-07,
"loss": 1.4533,
"step": 530
},
{
"epoch": 0.96,
"grad_norm": 11.5625,
"learning_rate": 5.5406087565471054e-08,
"loss": 1.5255,
"step": 540
},
{
"epoch": 0.9777777777777777,
"grad_norm": 8.8125,
"learning_rate": 1.8876494121959908e-08,
"loss": 1.4647,
"step": 550
},
{
"epoch": 0.9955555555555555,
"grad_norm": 12.3125,
"learning_rate": 1.5418296089358964e-09,
"loss": 1.4632,
"step": 560
}
],
"logging_steps": 10,
"max_steps": 563,
"num_input_tokens_seen": 0,
"num_train_epochs": 1,
"save_steps": 50,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": true
},
"attributes": {}
}
},
"total_flos": 3.869939812904755e+16,
"train_batch_size": 2,
"trial_name": null,
"trial_params": null
}