Files
AfriqueQwen-14B-multiturn_2/trainer_state.json
ModelHub XC 118553032a 初始化项目,由ModelHub XC社区提供模型
Model: israel/AfriqueQwen-14B-multiturn_2
Source: Original Platform
2026-09-07 18:08:16 +08:00

282 lines
7.2 KiB
JSON

{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 5.0,
"eval_steps": 500,
"global_step": 34265,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.1459321415541773,
"grad_norm": 2.5889114643259257,
"learning_rate": 2.915086081120514e-06,
"loss": 1.120510009765625,
"step": 1000
},
{
"epoch": 0.2918642831083546,
"grad_norm": 2.2954157454578574,
"learning_rate": 5.833090166326233e-06,
"loss": 0.919832763671875,
"step": 2000
},
{
"epoch": 0.43779642466253194,
"grad_norm": 1.9740216723771231,
"learning_rate": 8.751094251531953e-06,
"loss": 0.866156005859375,
"step": 3000
},
{
"epoch": 0.5837285662167092,
"grad_norm": 1.6025712306944424,
"learning_rate": 9.991513345767592e-06,
"loss": 0.8476719970703125,
"step": 4000
},
{
"epoch": 0.7296607077708865,
"grad_norm": 1.6016651354512892,
"learning_rate": 9.936020028278053e-06,
"loss": 0.8244376831054687,
"step": 5000
},
{
"epoch": 0.8755928493250639,
"grad_norm": 1.6615443914575065,
"learning_rate": 9.829343371836088e-06,
"loss": 0.79805810546875,
"step": 6000
},
{
"epoch": 1.021452024808464,
"grad_norm": 1.739839844907049,
"learning_rate": 9.672589544454328e-06,
"loss": 0.761253662109375,
"step": 7000
},
{
"epoch": 1.1673841663626414,
"grad_norm": 1.8214814374223849,
"learning_rate": 9.46738398205746e-06,
"loss": 0.614188232421875,
"step": 8000
},
{
"epoch": 1.3133163079168186,
"grad_norm": 1.5403816299740094,
"learning_rate": 9.215854533761766e-06,
"loss": 0.6142713623046875,
"step": 9000
},
{
"epoch": 1.459248449470996,
"grad_norm": 1.8486916036138623,
"learning_rate": 8.920609397454381e-06,
"loss": 0.61425341796875,
"step": 10000
},
{
"epoch": 1.6051805910251733,
"grad_norm": 1.9343055796796615,
"learning_rate": 8.584710074466158e-06,
"loss": 0.613021240234375,
"step": 11000
},
{
"epoch": 1.7511127325793505,
"grad_norm": 1.3591298836511538,
"learning_rate": 8.211639623780629e-06,
"loss": 0.6173475341796875,
"step": 12000
},
{
"epoch": 1.897044874133528,
"grad_norm": 1.4881517892702774,
"learning_rate": 7.805266544962458e-06,
"loss": 0.6191266479492188,
"step": 13000
},
{
"epoch": 2.042904049616928,
"grad_norm": 1.7095977461968044,
"learning_rate": 7.3698046643160645e-06,
"loss": 0.5406383056640625,
"step": 14000
},
{
"epoch": 2.1888361911711054,
"grad_norm": 1.3383769030634658,
"learning_rate": 6.909769440229038e-06,
"loss": 0.37209414672851565,
"step": 15000
},
{
"epoch": 2.334768332725283,
"grad_norm": 1.5984154081302522,
"learning_rate": 6.4299311407857035e-06,
"loss": 0.37489187622070314,
"step": 16000
},
{
"epoch": 2.48070047427946,
"grad_norm": 1.5178018437242096,
"learning_rate": 5.935265379168761e-06,
"loss": 0.3743317260742188,
"step": 17000
},
{
"epoch": 2.6266326158336373,
"grad_norm": 2.243985181618142,
"learning_rate": 5.430901519764892e-06,
"loss": 0.3719102172851563,
"step": 18000
},
{
"epoch": 2.7725647573878147,
"grad_norm": 1.817398251882934,
"learning_rate": 4.9220694899697185e-06,
"loss": 0.3770401611328125,
"step": 19000
},
{
"epoch": 2.918496898941992,
"grad_norm": 1.400063403289345,
"learning_rate": 4.414045549219315e-06,
"loss": 0.37033236694335936,
"step": 20000
},
{
"epoch": 3.064356074425392,
"grad_norm": 1.9568348297606881,
"learning_rate": 3.912097577588397e-06,
"loss": 0.28135635375976564,
"step": 21000
},
{
"epoch": 3.2102882159795696,
"grad_norm": 1.6677571993666531,
"learning_rate": 3.4214304512770823e-06,
"loss": 0.1744761199951172,
"step": 22000
},
{
"epoch": 3.356220357533747,
"grad_norm": 1.6030223311676182,
"learning_rate": 2.9471320714071095e-06,
"loss": 0.17265531921386718,
"step": 23000
},
{
"epoch": 3.502152499087924,
"grad_norm": 2.0719250695035543,
"learning_rate": 2.4941206057740675e-06,
"loss": 0.1734700469970703,
"step": 24000
},
{
"epoch": 3.6480846406421015,
"grad_norm": 1.715926337353297,
"learning_rate": 2.06709349062457e-06,
"loss": 0.17093397521972656,
"step": 25000
},
{
"epoch": 3.7940167821962785,
"grad_norm": 1.6180812645074687,
"learning_rate": 1.6704787212769829e-06,
"loss": 0.16509759521484374,
"step": 26000
},
{
"epoch": 3.939948923750456,
"grad_norm": 2.048625803274215,
"learning_rate": 1.3083889366705216e-06,
"loss": 0.16218829345703126,
"step": 27000
},
{
"epoch": 4.085808099233856,
"grad_norm": 0.9085706761456963,
"learning_rate": 9.845787739562829e-07,
"loss": 0.10715762329101562,
"step": 28000
},
{
"epoch": 4.231740240788033,
"grad_norm": 1.2437558088436274,
"learning_rate": 7.024059353355333e-07,
"loss": 0.0650710220336914,
"step": 29000
},
{
"epoch": 4.377672382342211,
"grad_norm": 1.0557746493951432,
"learning_rate": 4.64796370857008e-07,
"loss": 0.06532522583007813,
"step": 30000
},
{
"epoch": 4.523604523896388,
"grad_norm": 1.6388458174836673,
"learning_rate": 2.7421393820510846e-07,
"loss": 0.06470259857177735,
"step": 31000
},
{
"epoch": 4.669536665450566,
"grad_norm": 1.2105021817675277,
"learning_rate": 1.326348540874095e-07,
"loss": 0.06277722549438476,
"step": 32000
},
{
"epoch": 4.815468807004743,
"grad_norm": 1.0929519804389642,
"learning_rate": 4.152720214406214e-08,
"loss": 0.06431336212158204,
"step": 33000
},
{
"epoch": 4.96140094855892,
"grad_norm": 1.5036224042092285,
"learning_rate": 1.8357098688476238e-09,
"loss": 0.06442949676513672,
"step": 34000
},
{
"epoch": 5.0,
"step": 34265,
"total_flos": 879165899800576.0,
"train_loss": 0.4208107623917577,
"train_runtime": 110380.4687,
"train_samples_per_second": 2.483,
"train_steps_per_second": 0.31
}
],
"logging_steps": 1000,
"max_steps": 34265,
"num_input_tokens_seen": 0,
"num_train_epochs": 5,
"save_steps": 50000,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": true
},
"attributes": {}
}
},
"total_flos": 879165899800576.0,
"train_batch_size": 1,
"trial_name": null,
"trial_params": null
}