初始化项目,由ModelHub XC社区提供模型

Model: SeongryongJung/Qwen3-4B-Tooluse-GRPO-TR
Source: Original Platform
This commit is contained in:
ModelHub XC
2026-08-16 11:24:56 +08:00
commit 9af9efe7e1
37 changed files with 325910 additions and 0 deletions

View File

@@ -0,0 +1,33 @@
section,parameter,value,source
Run identity,Base model,Qwen/Qwen3-4B,queue/script override
Run identity,Dataset,Tool-use / tooluse,run_qwen3_generalization.sh
Run identity,Method,GRPO,run_qwen3_generalization.sh
Run identity,Config,baseline_grpo,run_qwen3_generalization.sh
Run identity,Experiment,qwen3gen-tooluse-GRPO-Qwen-Qwen3-4B-mbs8-train32-rollout8-lr1e-6-vllm0.8,run_qwen3_generalization.sh
Run identity,W&B run,run-20260702_145459-dx38nnn7,wandb
Data,Train file,datasets/tooluse/train.parquet,script override
Data,Validation file,datasets/tooluse/test.parquet,script override
Data,Train batch size,32,queue/script override
Data,Train max samples,3200,queue/script override
Schedule,Total training steps,100,queue/script override
Schedule,Validation before train,False,queue/script override
Schedule,Save frequency,10,queue/script override
Schedule,Validation frequency,10,queue/script override
Sequence,Max prompt length,2048,queue/script override
Sequence,Max response length,8192,queue/script override
Sequence,Max model length,10240,queue/script override
Rollout,Train rollout n,8,queue/script override
Rollout,Validation rollout n,16,queue/script override
Rollout,vLLM GPU memory utilization,0.8,queue/script override
Optimization,Learning rate,1e-6,GRPO method override
Optimization,Weight decay,0.01,script override
PPO/GRPO,PPO mini batch size,8,queue/script override
PPO/GRPO,Normalize GRPO advantages by std,False,baseline_grpo.yaml / script override
Rollout correction,Importance sampling mode,token,script override
Rollout correction,IS threshold,2.0,script override
Checkpoint/Logging,Checkpoint root,checkpoints/datasets/tooluse/qwen3gen-tooluse-GRPO-Qwen-Qwen3-4B-mbs8-train32-rollout8-lr1e-6-vllm0.8,script override
Checkpoint/Logging,Latest checkpointed iteration,100,latest_checkpointed_iteration.txt
Checkpoint/Logging,External actor archive,checkpoints/datasets/tooluse/qwen3gen-tooluse-GRPO-Qwen-Qwen3-4B-mbs8-train32-rollout8-lr1e-6-vllm0.8/_actor_archive,preserve_actor_checkpoints.py
Checkpoint/Logging,Logger,"console, wandb",ppo_trainer.yaml
PPO/GRPO,Policy loss mode,vanilla,method override
PPO/GRPO,Actor KL loss coef,0.0,method override
1 section parameter value source
2 Run identity Base model Qwen/Qwen3-4B queue/script override
3 Run identity Dataset Tool-use / tooluse run_qwen3_generalization.sh
4 Run identity Method GRPO run_qwen3_generalization.sh
5 Run identity Config baseline_grpo run_qwen3_generalization.sh
6 Run identity Experiment qwen3gen-tooluse-GRPO-Qwen-Qwen3-4B-mbs8-train32-rollout8-lr1e-6-vllm0.8 run_qwen3_generalization.sh
7 Run identity W&B run run-20260702_145459-dx38nnn7 wandb
8 Data Train file datasets/tooluse/train.parquet script override
9 Data Validation file datasets/tooluse/test.parquet script override
10 Data Train batch size 32 queue/script override
11 Data Train max samples 3200 queue/script override
12 Schedule Total training steps 100 queue/script override
13 Schedule Validation before train False queue/script override
14 Schedule Save frequency 10 queue/script override
15 Schedule Validation frequency 10 queue/script override
16 Sequence Max prompt length 2048 queue/script override
17 Sequence Max response length 8192 queue/script override
18 Sequence Max model length 10240 queue/script override
19 Rollout Train rollout n 8 queue/script override
20 Rollout Validation rollout n 16 queue/script override
21 Rollout vLLM GPU memory utilization 0.8 queue/script override
22 Optimization Learning rate 1e-6 GRPO method override
23 Optimization Weight decay 0.01 script override
24 PPO/GRPO PPO mini batch size 8 queue/script override
25 PPO/GRPO Normalize GRPO advantages by std False baseline_grpo.yaml / script override
26 Rollout correction Importance sampling mode token script override
27 Rollout correction IS threshold 2.0 script override
28 Checkpoint/Logging Checkpoint root checkpoints/datasets/tooluse/qwen3gen-tooluse-GRPO-Qwen-Qwen3-4B-mbs8-train32-rollout8-lr1e-6-vllm0.8 script override
29 Checkpoint/Logging Latest checkpointed iteration 100 latest_checkpointed_iteration.txt
30 Checkpoint/Logging External actor archive checkpoints/datasets/tooluse/qwen3gen-tooluse-GRPO-Qwen-Qwen3-4B-mbs8-train32-rollout8-lr1e-6-vllm0.8/_actor_archive preserve_actor_checkpoints.py
31 Checkpoint/Logging Logger console, wandb ppo_trainer.yaml
32 PPO/GRPO Policy loss mode vanilla method override
33 PPO/GRPO Actor KL loss coef 0.0 method override

27
results/summary.json Normal file
View File

@@ -0,0 +1,27 @@
{
"repo_id": "SeongryongJung/Qwen3-4B-Tooluse-GRPO-TR",
"output_dir": "/mnt/mole/SDPO/L2T/hf_upload_tr/Qwen3-4B-Tooluse-GRPO-TR",
"experiment": "qwen3gen-tooluse-GRPO-Qwen-Qwen3-4B-mbs8-train32-rollout8-lr1e-6-vllm0.8",
"best_step": 40,
"best_val_mean16": 0.6029411764705882,
"best_actor_dir": "/mnt/mole/SDPO/L2T/checkpoints/datasets/tooluse/qwen3gen-tooluse-GRPO-Qwen-Qwen3-4B-mbs8-train32-rollout8-lr1e-6-vllm0.8/_actor_archive/global_step_40/actor",
"final_step": 100,
"final_val_mean16": 0.5689338235294118,
"final_actor_dir": "/mnt/mole/SDPO/L2T/checkpoints/datasets/tooluse/qwen3gen-tooluse-GRPO-Qwen-Qwen3-4B-mbs8-train32-rollout8-lr1e-6-vllm0.8/global_step_100/actor",
"train_rows": 90,
"val_rows": 10,
"hf_model_files": [
"added_tokens.json",
"chat_template.jinja",
"config.json",
"generation_config.json",
"merges.txt",
"model.safetensors.index.json",
"special_tokens_map.json",
"tokenizer.json",
"tokenizer_config.json",
"vocab.json",
"model-00001-of-00002.safetensors",
"model-00002-of-00002.safetensors"
]
}

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:34070b4fc5932ea614982c3e664464876ddf22215fe056a1804bc20fb7fac39f
size 136037

1651
results/training_score.svg Normal file

File diff suppressed because it is too large Load Diff

After

Width:  |  Height:  |  Size: 45 KiB

View File

@@ -0,0 +1,91 @@
step,critic_score_mean
1,0.3125
2,0.24609375
3,0.34375
4,0.2734375
5,0.375
6,0.42578125
7,0.3828125
8,0.3046875
9,0.28515625
11,0.30078125
12,0.359375
13,0.26953125
14,0.25390625
15,0.5390625
16,0.3125
17,0.38671875
18,0.3046875
19,0.23828125
21,0.20703125
22,0.4453125
23,0.20703125
24,0.2890625
25,0.2109375
26,0.5234375
27,0.375
28,0.30078125
29,0.171875
31,0.421875
32,0.28125
33,0.2265625
34,0.359375
35,0.30078125
36,0.4140625
37,0.42578125
38,0.3828125
39,0.42578125
41,0.30859375
42,0.22265625
43,0.3125
44,0.265625
45,0.25390625
46,0.30859375
47,0.30078125
48,0.38671875
49,0.2109375
51,0.37109375
52,0.4375
53,0.36328125
54,0.3359375
55,0.4140625
56,0.36328125
57,0.4296875
58,0.37109375
59,0.4140625
61,0.3828125
62,0.33984375
63,0.3671875
64,0.38671875
65,0.30859375
66,0.421875
67,0.421875
68,0.46484375
69,0.5390625
71,0.3046875
72,0.38671875
73,0.33984375
74,0.421875
75,0.35546875
76,0.25
77,0.30078125
78,0.375
79,0.4140625
81,0.40625
82,0.44140625
83,0.25390625
84,0.46875
85,0.390625
86,0.40234375
87,0.4375
88,0.53515625
89,0.30859375
91,0.47265625
92,0.41015625
93,0.32421875
94,0.3125
95,0.30859375
96,0.42578125
97,0.33984375
98,0.359375
99,0.3828125
1 step critic_score_mean
2 1 0.3125
3 2 0.24609375
4 3 0.34375
5 4 0.2734375
6 5 0.375
7 6 0.42578125
8 7 0.3828125
9 8 0.3046875
10 9 0.28515625
11 11 0.30078125
12 12 0.359375
13 13 0.26953125
14 14 0.25390625
15 15 0.5390625
16 16 0.3125
17 17 0.38671875
18 18 0.3046875
19 19 0.23828125
20 21 0.20703125
21 22 0.4453125
22 23 0.20703125
23 24 0.2890625
24 25 0.2109375
25 26 0.5234375
26 27 0.375
27 28 0.30078125
28 29 0.171875
29 31 0.421875
30 32 0.28125
31 33 0.2265625
32 34 0.359375
33 35 0.30078125
34 36 0.4140625
35 37 0.42578125
36 38 0.3828125
37 39 0.42578125
38 41 0.30859375
39 42 0.22265625
40 43 0.3125
41 44 0.265625
42 45 0.25390625
43 46 0.30859375
44 47 0.30078125
45 48 0.38671875
46 49 0.2109375
47 51 0.37109375
48 52 0.4375
49 53 0.36328125
50 54 0.3359375
51 55 0.4140625
52 56 0.36328125
53 57 0.4296875
54 58 0.37109375
55 59 0.4140625
56 61 0.3828125
57 62 0.33984375
58 63 0.3671875
59 64 0.38671875
60 65 0.30859375
61 66 0.421875
62 67 0.421875
63 68 0.46484375
64 69 0.5390625
65 71 0.3046875
66 72 0.38671875
67 73 0.33984375
68 74 0.421875
69 75 0.35546875
70 76 0.25
71 77 0.30078125
72 78 0.375
73 79 0.4140625
74 81 0.40625
75 82 0.44140625
76 83 0.25390625
77 84 0.46875
78 85 0.390625
79 86 0.40234375
80 87 0.4375
81 88 0.53515625
82 89 0.30859375
83 91 0.47265625
84 92 0.41015625
85 93 0.32421875
86 94 0.3125
87 95 0.30859375
88 96 0.42578125
89 97 0.33984375
90 98 0.359375
91 99 0.3828125

View File

@@ -0,0 +1,11 @@
step,val_mean16
10,0.5965073529411765
20,0.5790441176470589
30,0.5909926470588235
40,0.6029411764705882
50,0.5946691176470589
60,0.5698529411764706
70,0.5496323529411765
80,0.5597426470588235
90,0.5643382352941176
100,0.5689338235294118
1 step val_mean16
2 10 0.5965073529411765
3 20 0.5790441176470589
4 30 0.5909926470588235
5 40 0.6029411764705882
6 50 0.5946691176470589
7 60 0.5698529411764706
8 70 0.5496323529411765
9 80 0.5597426470588235
10 90 0.5643382352941176
11 100 0.5689338235294118