初始化项目,由ModelHub XC社区提供模型

Model: SeongryongJung/Qwen3-8B-Physics-GRPO-TR
Source: Original Platform
This commit is contained in:
ModelHub XC
2026-08-07 04:05:18 +08:00
commit 90ed71fcdc
42 changed files with 316727 additions and 0 deletions

View File

@@ -0,0 +1,33 @@
section,parameter,value,source
Run identity,Base model,Qwen/Qwen3-8B,queue/script override
Run identity,Dataset,Physics / SciKnowEval physics,run_qwen3_generalization.sh
Run identity,Method,GRPO,run_qwen3_generalization.sh
Run identity,Config,baseline_grpo,run_qwen3_generalization.sh
Run identity,Experiment,qwen3gen-physics-GRPO-Qwen-Qwen3-8B-mbs8-train32-rollout8-lr1e-6-vllm0.8,run_qwen3_generalization.sh
Run identity,W&B run,run-20260703_032040-6ig3l55l,wandb
Data,Train file,datasets/sciknoweval/physics/train.parquet,script override
Data,Validation file,datasets/sciknoweval/physics/test.parquet,script override
Data,Train batch size,32,queue/script override
Data,Train max samples,3200,queue/script override
Schedule,Total training steps,100,queue/script override
Schedule,Validation before train,False,queue/script override
Schedule,Save frequency,10,queue/script override
Schedule,Validation frequency,10,queue/script override
Sequence,Max prompt length,2048,queue/script override
Sequence,Max response length,8192,queue/script override
Sequence,Max model length,10240,queue/script override
Rollout,Train rollout n,8,queue/script override
Rollout,Validation rollout n,16,queue/script override
Rollout,vLLM GPU memory utilization,0.8,queue/script override
Optimization,Learning rate,1e-6,GRPO method override
Optimization,Weight decay,0.01,script override
PPO/GRPO,PPO mini batch size,8,queue/script override
PPO/GRPO,Normalize GRPO advantages by std,False,baseline_grpo.yaml / script override
Rollout correction,Importance sampling mode,token,script override
Rollout correction,IS threshold,2.0,script override
Checkpoint/Logging,Checkpoint root,checkpoints/datasets/sciknoweval/physics/qwen3gen-physics-GRPO-Qwen-Qwen3-8B-mbs8-train32-rollout8-lr1e-6-vllm0.8,script override
Checkpoint/Logging,Latest checkpointed iteration,100,latest_checkpointed_iteration.txt
Checkpoint/Logging,External actor archive,checkpoints/datasets/sciknoweval/physics/qwen3gen-physics-GRPO-Qwen-Qwen3-8B-mbs8-train32-rollout8-lr1e-6-vllm0.8/_actor_archive,preserve_actor_checkpoints.py
Checkpoint/Logging,Logger,"console, wandb",ppo_trainer.yaml
PPO/GRPO,Policy loss mode,vanilla,method override
PPO/GRPO,Actor KL loss coef,0.0,method override
1 section parameter value source
2 Run identity Base model Qwen/Qwen3-8B queue/script override
3 Run identity Dataset Physics / SciKnowEval physics run_qwen3_generalization.sh
4 Run identity Method GRPO run_qwen3_generalization.sh
5 Run identity Config baseline_grpo run_qwen3_generalization.sh
6 Run identity Experiment qwen3gen-physics-GRPO-Qwen-Qwen3-8B-mbs8-train32-rollout8-lr1e-6-vllm0.8 run_qwen3_generalization.sh
7 Run identity W&B run run-20260703_032040-6ig3l55l wandb
8 Data Train file datasets/sciknoweval/physics/train.parquet script override
9 Data Validation file datasets/sciknoweval/physics/test.parquet script override
10 Data Train batch size 32 queue/script override
11 Data Train max samples 3200 queue/script override
12 Schedule Total training steps 100 queue/script override
13 Schedule Validation before train False queue/script override
14 Schedule Save frequency 10 queue/script override
15 Schedule Validation frequency 10 queue/script override
16 Sequence Max prompt length 2048 queue/script override
17 Sequence Max response length 8192 queue/script override
18 Sequence Max model length 10240 queue/script override
19 Rollout Train rollout n 8 queue/script override
20 Rollout Validation rollout n 16 queue/script override
21 Rollout vLLM GPU memory utilization 0.8 queue/script override
22 Optimization Learning rate 1e-6 GRPO method override
23 Optimization Weight decay 0.01 script override
24 PPO/GRPO PPO mini batch size 8 queue/script override
25 PPO/GRPO Normalize GRPO advantages by std False baseline_grpo.yaml / script override
26 Rollout correction Importance sampling mode token script override
27 Rollout correction IS threshold 2.0 script override
28 Checkpoint/Logging Checkpoint root checkpoints/datasets/sciknoweval/physics/qwen3gen-physics-GRPO-Qwen-Qwen3-8B-mbs8-train32-rollout8-lr1e-6-vllm0.8 script override
29 Checkpoint/Logging Latest checkpointed iteration 100 latest_checkpointed_iteration.txt
30 Checkpoint/Logging External actor archive checkpoints/datasets/sciknoweval/physics/qwen3gen-physics-GRPO-Qwen-Qwen3-8B-mbs8-train32-rollout8-lr1e-6-vllm0.8/_actor_archive preserve_actor_checkpoints.py
31 Checkpoint/Logging Logger console, wandb ppo_trainer.yaml
32 PPO/GRPO Policy loss mode vanilla method override
33 PPO/GRPO Actor KL loss coef 0.0 method override

29
results/summary.json Normal file
View File

@@ -0,0 +1,29 @@
{
"repo_id": "SeongryongJung/Qwen3-8B-Physics-GRPO-TR",
"output_dir": "/mnt/mole/SDPO/L2T/hf_upload_tr/Qwen3-8B-Physics-GRPO-TR",
"experiment": "qwen3gen-physics-GRPO-Qwen-Qwen3-8B-mbs8-train32-rollout8-lr1e-6-vllm0.8",
"best_step": 100,
"best_val_mean16": 0.7296875,
"best_actor_dir": "/mnt/mole/SDPO/L2T/checkpoints/datasets/sciknoweval/physics/qwen3gen-physics-GRPO-Qwen-Qwen3-8B-mbs8-train32-rollout8-lr1e-6-vllm0.8/global_step_100/actor",
"final_step": 100,
"final_val_mean16": 0.7296875,
"final_actor_dir": "/mnt/mole/SDPO/L2T/checkpoints/datasets/sciknoweval/physics/qwen3gen-physics-GRPO-Qwen-Qwen3-8B-mbs8-train32-rollout8-lr1e-6-vllm0.8/global_step_100/actor",
"train_rows": 90,
"val_rows": 10,
"hf_model_files": [
"added_tokens.json",
"chat_template.jinja",
"config.json",
"generation_config.json",
"merges.txt",
"model.safetensors.index.json",
"special_tokens_map.json",
"tokenizer.json",
"tokenizer_config.json",
"vocab.json",
"model-00001-of-00004.safetensors",
"model-00002-of-00004.safetensors",
"model-00003-of-00004.safetensors",
"model-00004-of-00004.safetensors"
]
}

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:453f8ccb73768a3b3ae4b970f7f0b84df4b8d760e9ac177c35f17270c51db265
size 129239

1591
results/training_score.svg Normal file

File diff suppressed because it is too large Load Diff

After

Width:  |  Height:  |  Size: 43 KiB

View File

@@ -0,0 +1,91 @@
step,critic_score_mean
1,0.6015625
2,0.5859375
3,0.52734375
4,0.63671875
5,0.73046875
6,0.65625
7,0.75390625
8,0.7265625
9,0.58984375
11,0.62890625
12,0.5234375
13,0.5703125
14,0.64453125
15,0.69140625
16,0.765625
17,0.70703125
18,0.41015625
19,0.71875
21,0.6640625
22,0.71484375
23,0.5
24,0.74609375
25,0.71875
26,0.7109375
27,0.796875
28,0.6171875
29,0.70703125
31,0.62890625
32,0.73828125
33,0.70703125
34,0.6796875
35,0.75390625
36,0.578125
37,0.6640625
38,0.66015625
39,0.87890625
41,0.70703125
42,0.6171875
43,0.74609375
44,0.66796875
45,0.66796875
46,0.7890625
47,0.75
48,0.765625
49,0.859375
51,0.7109375
52,0.71484375
53,0.73828125
54,0.76171875
55,0.84765625
56,0.6484375
57,0.734375
58,0.64453125
59,0.71875
61,0.72265625
62,0.65625
63,0.7109375
64,0.6875
65,0.8671875
66,0.765625
67,0.8046875
68,0.734375
69,0.80859375
71,0.82421875
72,0.76953125
73,0.66015625
74,0.82421875
75,0.85546875
76,0.6953125
77,0.72265625
78,0.80859375
79,0.82421875
81,0.734375
82,0.76953125
83,0.6953125
84,0.68359375
85,0.73828125
86,0.77734375
87,0.76953125
88,0.72265625
89,0.7421875
91,0.81640625
92,0.73046875
93,0.75
94,0.8203125
95,0.6328125
96,0.7421875
97,0.734375
98,0.79296875
99,0.84765625
1 step critic_score_mean
2 1 0.6015625
3 2 0.5859375
4 3 0.52734375
5 4 0.63671875
6 5 0.73046875
7 6 0.65625
8 7 0.75390625
9 8 0.7265625
10 9 0.58984375
11 11 0.62890625
12 12 0.5234375
13 13 0.5703125
14 14 0.64453125
15 15 0.69140625
16 16 0.765625
17 17 0.70703125
18 18 0.41015625
19 19 0.71875
20 21 0.6640625
21 22 0.71484375
22 23 0.5
23 24 0.74609375
24 25 0.71875
25 26 0.7109375
26 27 0.796875
27 28 0.6171875
28 29 0.70703125
29 31 0.62890625
30 32 0.73828125
31 33 0.70703125
32 34 0.6796875
33 35 0.75390625
34 36 0.578125
35 37 0.6640625
36 38 0.66015625
37 39 0.87890625
38 41 0.70703125
39 42 0.6171875
40 43 0.74609375
41 44 0.66796875
42 45 0.66796875
43 46 0.7890625
44 47 0.75
45 48 0.765625
46 49 0.859375
47 51 0.7109375
48 52 0.71484375
49 53 0.73828125
50 54 0.76171875
51 55 0.84765625
52 56 0.6484375
53 57 0.734375
54 58 0.64453125
55 59 0.71875
56 61 0.72265625
57 62 0.65625
58 63 0.7109375
59 64 0.6875
60 65 0.8671875
61 66 0.765625
62 67 0.8046875
63 68 0.734375
64 69 0.80859375
65 71 0.82421875
66 72 0.76953125
67 73 0.66015625
68 74 0.82421875
69 75 0.85546875
70 76 0.6953125
71 77 0.72265625
72 78 0.80859375
73 79 0.82421875
74 81 0.734375
75 82 0.76953125
76 83 0.6953125
77 84 0.68359375
78 85 0.73828125
79 86 0.77734375
80 87 0.76953125
81 88 0.72265625
82 89 0.7421875
83 91 0.81640625
84 92 0.73046875
85 93 0.75
86 94 0.8203125
87 95 0.6328125
88 96 0.7421875
89 97 0.734375
90 98 0.79296875
91 99 0.84765625

View File

@@ -0,0 +1,11 @@
step,val_mean16
10,0.58359375
20,0.59921875
30,0.6046875
40,0.62109375
50,0.65390625
60,0.671875
70,0.68125
80,0.7125
90,0.72265625
100,0.7296875
1 step val_mean16
2 10 0.58359375
3 20 0.59921875
4 30 0.6046875
5 40 0.62109375
6 50 0.65390625
7 60 0.671875
8 70 0.68125
9 80 0.7125
10 90 0.72265625
11 100 0.7296875