初始化项目,由ModelHub XC社区提供模型

Model: SeongryongJung/Qwen3-4B-Material-GRPO-TR
Source: Original Platform
This commit is contained in:
ModelHub XC
2026-08-04 08:21:18 +08:00
commit 5a9933097b
37 changed files with 326337 additions and 0 deletions

View File

@@ -0,0 +1,33 @@
section,parameter,value,source
Run identity,Base model,Qwen/Qwen3-4B,queue/script override
Run identity,Dataset,Material / SciKnowEval material,run_qwen3_generalization.sh
Run identity,Method,GRPO,run_qwen3_generalization.sh
Run identity,Config,baseline_grpo,run_qwen3_generalization.sh
Run identity,Experiment,qwen3gen-material-GRPO-Qwen-Qwen3-4B-mbs8-train32-rollout8-lr1e-6-vllm0.8,run_qwen3_generalization.sh
Run identity,W&B run,run-20260702_125526-lnzvk3fv,wandb
Data,Train file,datasets/sciknoweval/material/train.parquet,script override
Data,Validation file,datasets/sciknoweval/material/test.parquet,script override
Data,Train batch size,32,queue/script override
Data,Train max samples,3200,queue/script override
Schedule,Total training steps,100,queue/script override
Schedule,Validation before train,False,queue/script override
Schedule,Save frequency,10,queue/script override
Schedule,Validation frequency,10,queue/script override
Sequence,Max prompt length,2048,queue/script override
Sequence,Max response length,8192,queue/script override
Sequence,Max model length,10240,queue/script override
Rollout,Train rollout n,8,queue/script override
Rollout,Validation rollout n,16,queue/script override
Rollout,vLLM GPU memory utilization,0.8,queue/script override
Optimization,Learning rate,1e-6,GRPO method override
Optimization,Weight decay,0.01,script override
PPO/GRPO,PPO mini batch size,8,queue/script override
PPO/GRPO,Normalize GRPO advantages by std,False,baseline_grpo.yaml / script override
Rollout correction,Importance sampling mode,token,script override
Rollout correction,IS threshold,2.0,script override
Checkpoint/Logging,Checkpoint root,checkpoints/datasets/sciknoweval/material/qwen3gen-material-GRPO-Qwen-Qwen3-4B-mbs8-train32-rollout8-lr1e-6-vllm0.8,script override
Checkpoint/Logging,Latest checkpointed iteration,100,latest_checkpointed_iteration.txt
Checkpoint/Logging,External actor archive,checkpoints/datasets/sciknoweval/material/qwen3gen-material-GRPO-Qwen-Qwen3-4B-mbs8-train32-rollout8-lr1e-6-vllm0.8/_actor_archive,preserve_actor_checkpoints.py
Checkpoint/Logging,Logger,"console, wandb",ppo_trainer.yaml
PPO/GRPO,Policy loss mode,vanilla,method override
PPO/GRPO,Actor KL loss coef,0.0,method override
1 section parameter value source
2 Run identity Base model Qwen/Qwen3-4B queue/script override
3 Run identity Dataset Material / SciKnowEval material run_qwen3_generalization.sh
4 Run identity Method GRPO run_qwen3_generalization.sh
5 Run identity Config baseline_grpo run_qwen3_generalization.sh
6 Run identity Experiment qwen3gen-material-GRPO-Qwen-Qwen3-4B-mbs8-train32-rollout8-lr1e-6-vllm0.8 run_qwen3_generalization.sh
7 Run identity W&B run run-20260702_125526-lnzvk3fv wandb
8 Data Train file datasets/sciknoweval/material/train.parquet script override
9 Data Validation file datasets/sciknoweval/material/test.parquet script override
10 Data Train batch size 32 queue/script override
11 Data Train max samples 3200 queue/script override
12 Schedule Total training steps 100 queue/script override
13 Schedule Validation before train False queue/script override
14 Schedule Save frequency 10 queue/script override
15 Schedule Validation frequency 10 queue/script override
16 Sequence Max prompt length 2048 queue/script override
17 Sequence Max response length 8192 queue/script override
18 Sequence Max model length 10240 queue/script override
19 Rollout Train rollout n 8 queue/script override
20 Rollout Validation rollout n 16 queue/script override
21 Rollout vLLM GPU memory utilization 0.8 queue/script override
22 Optimization Learning rate 1e-6 GRPO method override
23 Optimization Weight decay 0.01 script override
24 PPO/GRPO PPO mini batch size 8 queue/script override
25 PPO/GRPO Normalize GRPO advantages by std False baseline_grpo.yaml / script override
26 Rollout correction Importance sampling mode token script override
27 Rollout correction IS threshold 2.0 script override
28 Checkpoint/Logging Checkpoint root checkpoints/datasets/sciknoweval/material/qwen3gen-material-GRPO-Qwen-Qwen3-4B-mbs8-train32-rollout8-lr1e-6-vllm0.8 script override
29 Checkpoint/Logging Latest checkpointed iteration 100 latest_checkpointed_iteration.txt
30 Checkpoint/Logging External actor archive checkpoints/datasets/sciknoweval/material/qwen3gen-material-GRPO-Qwen-Qwen3-4B-mbs8-train32-rollout8-lr1e-6-vllm0.8/_actor_archive preserve_actor_checkpoints.py
31 Checkpoint/Logging Logger console, wandb ppo_trainer.yaml
32 PPO/GRPO Policy loss mode vanilla method override
33 PPO/GRPO Actor KL loss coef 0.0 method override

27
results/summary.json Normal file
View File

@@ -0,0 +1,27 @@
{
"repo_id": "SeongryongJung/Qwen3-4B-Material-GRPO-TR",
"output_dir": "/mnt/mole/SDPO/L2T/hf_upload_tr/Qwen3-4B-Material-GRPO-TR",
"experiment": "qwen3gen-material-GRPO-Qwen-Qwen3-4B-mbs8-train32-rollout8-lr1e-6-vllm0.8",
"best_step": 60,
"best_val_mean16": 0.7659574468085106,
"best_actor_dir": "/mnt/mole/SDPO/L2T/checkpoints/datasets/sciknoweval/material/qwen3gen-material-GRPO-Qwen-Qwen3-4B-mbs8-train32-rollout8-lr1e-6-vllm0.8/_actor_archive/global_step_60/actor",
"final_step": 100,
"final_val_mean16": 0.7626329787234043,
"final_actor_dir": "/mnt/mole/SDPO/L2T/checkpoints/datasets/sciknoweval/material/qwen3gen-material-GRPO-Qwen-Qwen3-4B-mbs8-train32-rollout8-lr1e-6-vllm0.8/global_step_100/actor",
"train_rows": 90,
"val_rows": 10,
"hf_model_files": [
"added_tokens.json",
"chat_template.jinja",
"config.json",
"generation_config.json",
"merges.txt",
"model.safetensors.index.json",
"special_tokens_map.json",
"tokenizer.json",
"tokenizer_config.json",
"vocab.json",
"model-00001-of-00002.safetensors",
"model-00002-of-00002.safetensors"
]
}

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:9b7a0f03d7ae334cc4f31318502f2c08d5134f1349afab19faf7f278572303e2
size 150554

1723
results/training_score.svg Normal file

File diff suppressed because it is too large Load Diff

After

Width:  |  Height:  |  Size: 47 KiB

View File

@@ -0,0 +1,91 @@
step,critic_score_mean
1,0.5625
2,0.53515625
3,0.62109375
4,0.578125
5,0.56640625
6,0.640625
7,0.6484375
8,0.58984375
9,0.6328125
11,0.4921875
12,0.58984375
13,0.54296875
14,0.76953125
15,0.66015625
16,0.6953125
17,0.5703125
18,0.5546875
19,0.63671875
21,0.546875
22,0.75
23,0.80859375
24,0.89453125
25,0.6796875
26,0.65234375
27,0.8203125
28,0.6171875
29,0.68359375
31,0.63671875
32,0.6640625
33,0.8203125
34,0.7890625
35,0.7734375
36,0.671875
37,0.86328125
38,0.6171875
39,0.77734375
41,0.5859375
42,0.68359375
43,0.5859375
44,0.73828125
45,0.78515625
46,0.78125
47,0.69921875
48,0.83984375
49,0.640625
51,0.703125
52,0.76953125
53,0.6328125
54,0.5546875
55,0.703125
56,0.81640625
57,0.7578125
58,0.828125
59,0.65234375
61,0.81640625
62,0.8125
63,0.71875
64,0.74609375
65,0.60546875
66,0.70703125
67,0.78515625
68,0.81640625
69,0.83203125
71,0.74609375
72,0.7734375
73,0.70703125
74,0.83203125
75,0.69140625
76,0.734375
77,0.5546875
78,0.609375
79,0.7734375
81,0.71875
82,0.7265625
83,0.6875
84,0.7421875
85,0.7421875
86,0.84765625
87,0.625
88,0.6640625
89,0.74609375
91,0.80859375
92,0.7109375
93,0.78515625
94,0.73828125
95,0.76171875
96,0.58984375
97,0.85546875
98,0.8828125
99,0.71875
1 step critic_score_mean
2 1 0.5625
3 2 0.53515625
4 3 0.62109375
5 4 0.578125
6 5 0.56640625
7 6 0.640625
8 7 0.6484375
9 8 0.58984375
10 9 0.6328125
11 11 0.4921875
12 12 0.58984375
13 13 0.54296875
14 14 0.76953125
15 15 0.66015625
16 16 0.6953125
17 17 0.5703125
18 18 0.5546875
19 19 0.63671875
20 21 0.546875
21 22 0.75
22 23 0.80859375
23 24 0.89453125
24 25 0.6796875
25 26 0.65234375
26 27 0.8203125
27 28 0.6171875
28 29 0.68359375
29 31 0.63671875
30 32 0.6640625
31 33 0.8203125
32 34 0.7890625
33 35 0.7734375
34 36 0.671875
35 37 0.86328125
36 38 0.6171875
37 39 0.77734375
38 41 0.5859375
39 42 0.68359375
40 43 0.5859375
41 44 0.73828125
42 45 0.78515625
43 46 0.78125
44 47 0.69921875
45 48 0.83984375
46 49 0.640625
47 51 0.703125
48 52 0.76953125
49 53 0.6328125
50 54 0.5546875
51 55 0.703125
52 56 0.81640625
53 57 0.7578125
54 58 0.828125
55 59 0.65234375
56 61 0.81640625
57 62 0.8125
58 63 0.71875
59 64 0.74609375
60 65 0.60546875
61 66 0.70703125
62 67 0.78515625
63 68 0.81640625
64 69 0.83203125
65 71 0.74609375
66 72 0.7734375
67 73 0.70703125
68 74 0.83203125
69 75 0.69140625
70 76 0.734375
71 77 0.5546875
72 78 0.609375
73 79 0.7734375
74 81 0.71875
75 82 0.7265625
76 83 0.6875
77 84 0.7421875
78 85 0.7421875
79 86 0.84765625
80 87 0.625
81 88 0.6640625
82 89 0.74609375
83 91 0.80859375
84 92 0.7109375
85 93 0.78515625
86 94 0.73828125
87 95 0.76171875
88 96 0.58984375
89 97 0.85546875
90 98 0.8828125
91 99 0.71875

View File

@@ -0,0 +1,11 @@
step,val_mean16
10,0.6688829787234043
20,0.6954787234042553
30,0.7167553191489362
40,0.7393617021276596
50,0.754654255319149
60,0.7659574468085106
70,0.75
80,0.7506648936170213
90,0.7566489361702128
100,0.7626329787234043
1 step val_mean16
2 10 0.6688829787234043
3 20 0.6954787234042553
4 30 0.7167553191489362
5 40 0.7393617021276596
6 50 0.754654255319149
7 60 0.7659574468085106
8 70 0.75
9 80 0.7506648936170213
10 90 0.7566489361702128
11 100 0.7626329787234043