Files
nilechat-eg-lora-heuristic/checkpoint-350/trainer_state.json
ModelHub XC dbeeed6915 初始化项目,由ModelHub XC社区提供模型
Model: Henry236/nilechat-eg-lora-heuristic
Source: Original Platform
2026-09-02 01:28:17 +08:00

154 lines
3.6 KiB
JSON

{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 0.9009009009009009,
"eval_steps": 500,
"global_step": 350,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.05148005148005148,
"grad_norm": 3.7984347343444824,
"learning_rate": 2.435897435897436e-06,
"loss": 3.0721,
"step": 20
},
{
"epoch": 0.10296010296010295,
"grad_norm": 2.0532360076904297,
"learning_rate": 5e-06,
"loss": 2.7105,
"step": 40
},
{
"epoch": 0.15444015444015444,
"grad_norm": 1.889143466949463,
"learning_rate": 7.564102564102564e-06,
"loss": 2.397,
"step": 60
},
{
"epoch": 0.2059202059202059,
"grad_norm": 2.206676959991455,
"learning_rate": 9.999949644960027e-06,
"loss": 2.1587,
"step": 80
},
{
"epoch": 0.2574002574002574,
"grad_norm": 2.4184727668762207,
"learning_rate": 9.9778098230154e-06,
"loss": 1.9613,
"step": 100
},
{
"epoch": 0.3088803088803089,
"grad_norm": 2.1078503131866455,
"learning_rate": 9.915591603280632e-06,
"loss": 1.9187,
"step": 120
},
{
"epoch": 0.36036036036036034,
"grad_norm": 2.228259801864624,
"learning_rate": 9.813795930277306e-06,
"loss": 1.863,
"step": 140
},
{
"epoch": 0.4118404118404118,
"grad_norm": 3.990351438522339,
"learning_rate": 9.673242402909555e-06,
"loss": 1.7983,
"step": 160
},
{
"epoch": 0.46332046332046334,
"grad_norm": 2.6102797985076904,
"learning_rate": 9.495062675535614e-06,
"loss": 1.7845,
"step": 180
},
{
"epoch": 0.5148005148005148,
"grad_norm": 2.4415857791900635,
"learning_rate": 9.280691346552308e-06,
"loss": 1.7541,
"step": 200
},
{
"epoch": 0.5662805662805663,
"grad_norm": 2.875765800476074,
"learning_rate": 9.031854407852317e-06,
"loss": 1.7665,
"step": 220
},
{
"epoch": 0.6177606177606177,
"grad_norm": 2.55922269821167,
"learning_rate": 8.750555348152299e-06,
"loss": 1.7272,
"step": 240
},
{
"epoch": 0.6692406692406693,
"grad_norm": 3.4958205223083496,
"learning_rate": 8.439059022079789e-06,
"loss": 1.7349,
"step": 260
},
{
"epoch": 0.7207207207207207,
"grad_norm": 2.582611560821533,
"learning_rate": 8.099873414895453e-06,
"loss": 1.7273,
"step": 280
},
{
"epoch": 0.7722007722007722,
"grad_norm": 2.274104595184326,
"learning_rate": 7.73572944967043e-06,
"loss": 1.695,
"step": 300
},
{
"epoch": 0.8236808236808236,
"grad_norm": 2.8610410690307617,
"learning_rate": 7.3495589994995274e-06,
"loss": 1.663,
"step": 320
},
{
"epoch": 0.8751608751608752,
"grad_norm": 2.4936399459838867,
"learning_rate": 6.944471281782975e-06,
"loss": 1.6462,
"step": 340
}
],
"logging_steps": 20,
"max_steps": 778,
"num_input_tokens_seen": 0,
"num_train_epochs": 2,
"save_steps": 50,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 2.5584058931429376e+16,
"train_batch_size": 4,
"trial_name": null,
"trial_params": null
}