Files
Qwen3-8B-Physics-GRPO-TR/artifacts/config.yaml
ModelHub XC 90ed71fcdc 初始化项目,由ModelHub XC社区提供模型
Model: SeongryongJung/Qwen3-8B-Physics-GRPO-TR
Source: Original Platform
2026-08-07 04:05:18 +08:00

856 lines
31 KiB
YAML

_wandb:
value:
cli_version: 0.23.1
e:
1n1ucg63toheyodgdd2dcwkutcbcxhn8:
args:
- --node-ip-address=198.19.44.212
- --node-manager-port=36871
- --object-store-name=/tmp/ray_q3g_physics_grpo_Qwen3_8B/session_2026-07-03_03-18-52_070071_2279047/sockets/plasma_store
- --raylet-name=/tmp/ray_q3g_physics_grpo_Qwen3_8B/session_2026-07-03_03-18-52_070071_2279047/sockets/raylet
- --redis-address=None
- --metrics-agent-port=56405
- --logging-rotate-bytes=536870912
- --logging-rotate-backup-count=5
- --runtime-env-agent-port=64277
- --gcs-address=198.19.44.212:64574
- --session-name=session_2026-07-03_03-18-52_070071_2279047
- --temp-dir=/tmp/ray_q3g_physics_grpo_Qwen3_8B
- --webui=
- --cluster-id=d7576956015edcd024cf2ca706acce8be3fb6519bb7b0c87bfccae89
- --startup-token=32
- --worker-launch-time-ms=1783048735496
- --node-id=e88e44745be3f8c4449808648428760387c2aa53671532e80c9f13b3
- --runtime-env-hash=1652680156
codePath: .venv/lib/python3.12/site-packages/ray/_private/workers/default_worker.py
codePathLocal: .venv/lib/python3.12/site-packages/ray/_private/workers/default_worker.py
cpu_count: 64
cpu_count_logical: 128
cudaVersion: "13.0"
disk:
/:
total: "46086056050688"
used: "25787742543872"
email: jungsr1116@cau.ac.kr
executable: /mnt/mole/SDPO/L2T/.venv/bin/python
git:
commit: ee6a667700697069b93db636aa937f970c1fb57a
remote: https://github.com/jungseongryong/L2T.git
gpu: NVIDIA H200
gpu_count: 8
gpu_nvidia:
- architecture: Hopper
cudaCores: 16896
memoryTotal: "150754820096"
name: NVIDIA H200
uuid: GPU-c0451621-891c-7169-976d-71d81958db34
- architecture: Hopper
cudaCores: 16896
memoryTotal: "150754820096"
name: NVIDIA H200
uuid: GPU-e2a29dfa-7dad-6cd7-fb05-79d0060f27f0
- architecture: Hopper
cudaCores: 16896
memoryTotal: "150754820096"
name: NVIDIA H200
uuid: GPU-d69bb741-f2a7-a375-7f56-5c0d09d797b2
- architecture: Hopper
cudaCores: 16896
memoryTotal: "150754820096"
name: NVIDIA H200
uuid: GPU-1bbbea1e-7541-3126-1bfc-51b43a67577e
- architecture: Hopper
cudaCores: 16896
memoryTotal: "150754820096"
name: NVIDIA H200
uuid: GPU-6e0055af-452d-3f46-ce21-0c71e347a608
- architecture: Hopper
cudaCores: 16896
memoryTotal: "150754820096"
name: NVIDIA H200
uuid: GPU-317fdb5b-b280-71d8-398e-a843ad897437
- architecture: Hopper
cudaCores: 16896
memoryTotal: "150754820096"
name: NVIDIA H200
uuid: GPU-f19d8161-d2dd-49a1-dcb8-5c341b5e4514
- architecture: Hopper
cudaCores: 16896
memoryTotal: "150754820096"
name: NVIDIA H200
uuid: GPU-961f5ebf-4bf8-6855-828b-c065af57b87c
host: mole-workspace-rw
memory:
total: "2163980390400"
os: Linux-6.8.0-71-generic-x86_64-with-glibc2.36
program: /mnt/mole/SDPO/L2T/.venv/lib/python3.12/site-packages/ray/_private/workers/default_worker.py
python: CPython 3.12.13
root: /mnt/mole/SDPO/L2T
startedAt: "2026-07-03T03:20:40.327312Z"
writerId: 1n1ucg63toheyodgdd2dcwkutcbcxhn8
m: []
python_version: 3.12.13
t:
"1":
- 1
- 11
- 30
- 41
- 49
- 50
- 51
- 71
- 95
- 98
- 105
"2":
- 1
- 11
- 30
- 41
- 49
- 50
- 51
- 71
- 95
- 98
- 105
"3":
- 2
- 13
- 16
- 61
"4": 3.12.13
"5": 0.23.1
"6": 4.57.1
"12": 0.23.1
"13": linux-x86_64
actor_rollout_ref:
value:
actor:
_target_: verl.workers.config.FSDPActorConfig
calculate_entropy: false
calculate_sum_pi_squared: false
checkpoint:
_target_: verl.trainer.config.CheckpointConfig
async_save: false
load_contents:
- model
- optimizer
- extra
save_contents:
- model
- optimizer
- extra
clip_ratio: 0.2
clip_ratio_c: 3
clip_ratio_high: 0.28
clip_ratio_low: 0.2
data_loader_seed: 42
entropy_checkpointing: false
entropy_coeff: 0
entropy_from_logits_with_chunking: false
freeze_vision_tower: false
fsdp_config:
_target_: verl.workers.config.FSDPEngineConfig
dtype: bfloat16
entropy_checkpointing: false
entropy_from_logits_with_chunking: false
forward_only: false
forward_prefetch: false
fsdp_size: -1
full_determinism: false
model_dtype: fp32
offload_policy: false
optimizer_offload: false
param_offload: false
reshard_after_forward: true
seed: 42
strategy: fsdp
ulysses_sequence_parallel_size: 1
use_orig_params: false
use_torch_compile: true
wrap_policy:
min_num_params: 0
grad_clip: 1
kl_loss_coef: 0
kl_loss_type: low_var_kl
loss_agg_mode: token-mean
loss_scale_factor: null
optim:
_target_: verl.workers.config.FSDPOptimizerConfig
betas:
- 0.9
- 0.999
clip_grad: 1
lr: 1e-06
lr_scheduler_type: constant
lr_warmup_steps: 10
lr_warmup_steps_ratio: 0
min_lr_ratio: 0
num_cycles: 0.5
optimizer: AdamW
optimizer_impl: torch.optim
override_optimizer_config: null
total_training_steps: 100
warmup_style: null
weight_decay: 0.01
policy_loss:
_target_: verl.workers.config.PolicyLossConfig
clip_cov_lb: 1
clip_cov_ratio: 0.0002
clip_cov_ub: 5
kl_cov_ratio: 0.0002
loss_mode: vanilla
ppo_kl_coef: 0.1
ppo_epochs: 1
ppo_max_token_len_per_gpu: 10240
ppo_micro_batch_size: null
ppo_micro_batch_size_per_gpu: 1
ppo_mini_batch_size: 8
profiler:
_target_: verl.utils.profiler.ProfilerConfig
all_ranks: false
enable: false
ranks: []
save_path: outputs/profile
tool: null
tool_config:
npu:
_target_: verl.utils.profiler.config.NPUToolConfig
analysis: true
contents: []
discrete: false
level: level0
nsys:
_target_: verl.utils.profiler.config.NsightToolConfig
discrete: false
torch:
_target_: verl.utils.profiler.config.TorchProfilerToolConfig
step_end: null
step_start: 0
torch_memory:
_target_: verl.utils.profiler.config.TorchMemoryToolConfig
stack_depth: 32
trace_alloc_max_entries: 100000
rollout_n: 8
router_replay:
_target_: verl.workers.config.RouterReplayConfig
mode: disabled
record_file: null
replay_file: null
self_distillation:
_target_: verl.workers.config.SelfDistillationConfig
alpha: 0.5
distillation_add_tail: true
distillation_topk: 100
dont_reprompt_on_self_success: true
environment_feedback_only_without_solution: true
evolving_teacher:
_target_: verl.workers.config.EvolvingTeacherConfig
enable: false
loss_weight: 0
mask: all
feedback_template: |4-
The following is feedback from your unsuccessful earlier attempt:
{feedback_raw}
full_logit_distillation: true
include_environment_feedback: true
is_clip: 2
max_reprompt_len: 22528
remove_thinking_from_demonstration: false
reprompt_template: |-
{prompt}{solution}{feedback}
Correctly solve the original question.
reprompt_truncation: right
solution_template: |4-
Correct solution:
{successful_previous_attempt}
srpo_dynamic_weighting: false
srpo_dynamic_weighting_temperature: 1
success_reward_threshold: 0.5
teacher_regularization: ema
teacher_update_rate: 0
token_reweight_decay_steps: null
token_reweight_eps_w: 0.2
token_reweight_lambda: 0.5
shuffle: false
strategy: fsdp
sum_pi_squared_checkpointing: false
tau_neg: 1.05
tau_pos: 1
ulysses_sequence_parallel_size: 1
use_dynamic_bsz: false
use_fused_kernels: false
use_kl_loss: false
use_prefix_grouper: false
use_remove_padding: true
use_torch_compile: true
hybrid_engine: true
model:
_target_: verl.workers.config.HFModelConfig
custom_chat_template: null
enable_activation_offload: false
enable_gradient_checkpointing: true
exclude_modules: null
external_lib: null
fused_kernel_options:
impl_backend: torch
hf_config_path: null
lora_adapter_path: null
lora_alpha: 16
lora_rank: 0
path: Qwen/Qwen3-8B
target_modules: all-linear
tiled_mlp:
enabled: false
num_shards: 4
tokenizer_path: null
trust_remote_code: true
use_fused_kernels: false
use_liger: false
use_remove_padding: true
use_shm: false
nccl_timeout: 600
ref:
_target_: verl.workers.config.FSDPActorConfig
entropy_checkpointing: false
entropy_from_logits_with_chunking: false
fsdp_config:
_target_: verl.workers.config.FSDPEngineConfig
dtype: bfloat16
entropy_checkpointing: false
entropy_from_logits_with_chunking: false
forward_only: true
forward_prefetch: false
fsdp_size: -1
full_determinism: false
model_dtype: fp32
offload_policy: false
optimizer_offload: false
param_offload: false
reshard_after_forward: true
seed: 42
strategy: fsdp
ulysses_sequence_parallel_size: 1
use_orig_params: false
use_torch_compile: true
wrap_policy:
min_num_params: 0
log_prob_max_token_len_per_gpu: 10240
log_prob_micro_batch_size: null
log_prob_micro_batch_size_per_gpu: 1
log_prob_use_dynamic_bsz: false
profiler:
_target_: verl.utils.profiler.ProfilerConfig
all_ranks: false
enable: false
ranks: []
save_path: outputs/profile
tool: null
tool_config:
npu:
_target_: verl.utils.profiler.config.NPUToolConfig
analysis: true
contents: []
discrete: false
level: level0
nsys:
_target_: verl.utils.profiler.config.NsightToolConfig
discrete: false
torch:
_target_: verl.utils.profiler.config.TorchProfilerToolConfig
step_end: null
step_start: 0
torch_memory:
_target_: verl.utils.profiler.config.TorchMemoryToolConfig
stack_depth: 32
trace_alloc_max_entries: 100000
rollout_n: 8
router_replay:
_target_: verl.workers.config.RouterReplayConfig
mode: disabled
record_file: null
replay_file: null
strategy: fsdp
ulysses_sequence_parallel_size: 1
use_torch_compile: true
rollout:
_target_: verl.workers.config.RolloutConfig
agent:
_target_: verl.workers.config.AgentLoopConfig
agent_loop_config_path: null
custom_async_server:
_target_: verl.workers.config.CustomAsyncServerConfig
name: null
path: null
default_agent_loop: single_turn_agent
num_workers: 8
calculate_log_probs: true
cudagraph_capture_sizes: null
data_parallel_size: 1
disable_log_stats: true
do_sample: true
dtype: bfloat16
enable_chunked_prefill: true
enable_prefix_caching: true
enable_rollout_routing_replay: false
enforce_eager: false
expert_parallel_size: 1
free_cache_engine: true
gpu_memory_utilization: 0.8
ignore_eos: false
layered_summon: false
load_format: dummy
log_prob_max_token_len_per_gpu: 10240
log_prob_micro_batch_size: null
log_prob_micro_batch_size_per_gpu: 1
log_prob_use_dynamic_bsz: false
logprobs_mode: processed_logprobs
max_model_len: 10240
max_num_batched_tokens: 10240
max_num_seqs: 1024
mode: async
multi_stage_wake_up: false
multi_turn:
_target_: verl.workers.config.MultiTurnConfig
enable: false
format: hermes
interaction_config_path: null
max_assistant_turns: null
max_parallel_calls: 1
max_tool_response_length: 256
max_user_turns: null
num_repeat_rollouts: null
tokenization_sanity_check_mode: strict
tool_config_path: null
tool_response_truncate_side: middle
use_inference_chat_template: false
"n": 8
name: vllm
over_sample_rate: 0
pipeline_model_parallel_size: 1
profiler:
_target_: verl.utils.profiler.ProfilerConfig
all_ranks: false
enable: false
ranks: []
save_path: outputs/profile
tool: null
tool_config:
npu:
_target_: verl.utils.profiler.config.NPUToolConfig
analysis: true
contents: []
discrete: false
level: level0
nsys:
_target_: verl.utils.profiler.config.NsightToolConfig
discrete: false
torch:
_target_: verl.utils.profiler.config.TorchProfilerToolConfig
step_end: null
step_start: 0
torch_memory:
_target_: verl.utils.profiler.config.TorchMemoryToolConfig
stack_depth: 32
trace_alloc_max_entries: 100000
prometheus:
_target_: verl.workers.config.PrometheusConfig
enable: false
file: /tmp/ray/session_latest/metrics/prometheus/prometheus.yml
port: 9090
served_model_name: Qwen/Qwen3-8B
prompt_length: 2048
quantization: null
quantization_config_file: null
response_length: 8192
scheduling_policy: fcfs
skip_dump_dir: /tmp/rollout_dump
skip_rollout: false
skip_tokenizer_init: true
temperature: 1
tensor_model_parallel_size: 2
top_k: -1
top_p: 1
trace:
_target_: verl.workers.config.TraceConfig
backend: null
max_samples_per_step_per_worker: null
token2text: false
update_weights_bucket_megabytes: 512
val_kwargs:
_target_: verl.workers.config.SamplingConfig
do_sample: true
"n": 16
temperature: 0.6
top_k: -1
top_p: 0.95
algorithm:
value:
_target_: verl.trainer.config.AlgoConfig
adv_estimator: grpo
gamma: 1
kl_ctrl:
_target_: verl.trainer.config.KLControlConfig
horizon: 10000
kl_coef: 0.001
target_kl: 0.1
type: fixed
kl_penalty: kl
lam: 1
norm_adv_by_std_in_grpo: false
pf_ppo:
reweight_method: pow
weight_pow: 2
rollout_correction:
bypass_mode: false
loss_type: ppo_clip
rollout_is: token
rollout_is_batch_normalize: false
rollout_is_threshold: 2
rollout_rs: null
rollout_rs_threshold: null
use_kl_in_reward: false
use_pf_ppo: false
critic:
value:
_target_: verl.workers.config.FSDPCriticConfig
checkpoint:
_target_: verl.trainer.config.CheckpointConfig
async_save: false
load_contents:
- model
- optimizer
- extra
save_contents:
- model
- optimizer
- extra
cliprange_value: 0.5
data_loader_seed: 42
enable: null
forward_max_token_len_per_gpu: 32768
forward_micro_batch_size: null
forward_micro_batch_size_per_gpu: null
grad_clip: 1
loss_agg_mode: token-mean
model:
_target_: verl.workers.config.FSDPCriticModelCfg
enable_activation_offload: false
enable_gradient_checkpointing: true
external_lib: null
fsdp_config:
_target_: verl.workers.config.FSDPEngineConfig
dtype: bfloat16
entropy_checkpointing: false
entropy_from_logits_with_chunking: false
forward_only: false
forward_prefetch: false
fsdp_size: -1
full_determinism: false
model_dtype: fp32
offload_policy: false
optimizer_offload: false
param_offload: false
reshard_after_forward: true
seed: 42
strategy: fsdp
ulysses_sequence_parallel_size: 1
use_orig_params: false
use_torch_compile: true
wrap_policy:
min_num_params: 0
lora_alpha: 16
lora_rank: 0
path: Qwen/Qwen3-8B
target_modules: all-linear
tiled_mlp:
enabled: false
num_shards: 4
tokenizer_path: Qwen/Qwen3-8B
trust_remote_code: true
use_remove_padding: false
use_shm: false
optim:
_target_: verl.workers.config.FSDPOptimizerConfig
betas:
- 0.9
- 0.999
clip_grad: 1
lr: 1e-05
lr_scheduler_type: constant
lr_warmup_steps: -1
lr_warmup_steps_ratio: 0
min_lr_ratio: 0
num_cycles: 0.5
optimizer: AdamW
optimizer_impl: torch.optim
override_optimizer_config: null
total_training_steps: 100
warmup_style: null
weight_decay: 0.01
ppo_epochs: 1
ppo_max_token_len_per_gpu: 32768
ppo_micro_batch_size: null
ppo_micro_batch_size_per_gpu: null
ppo_mini_batch_size: 8
profiler:
_target_: verl.utils.profiler.ProfilerConfig
all_ranks: false
enable: false
ranks: []
save_path: outputs/profile
tool: null
tool_config:
npu:
_target_: verl.utils.profiler.config.NPUToolConfig
analysis: true
contents: []
discrete: false
level: level0
nsys:
_target_: verl.utils.profiler.config.NsightToolConfig
discrete: false
torch:
_target_: verl.utils.profiler.config.TorchProfilerToolConfig
step_end: null
step_start: 0
torch_memory:
_target_: verl.utils.profiler.config.TorchMemoryToolConfig
stack_depth: 32
trace_alloc_max_entries: 100000
rollout_n: 8
shuffle: false
strategy: fsdp
ulysses_sequence_parallel_size: 1
use_dynamic_bsz: false
custom_reward_function:
value:
name: compute_score
path: /mnt/mole/SDPO/L2T/verl/utils/reward_score/feedback/__init__.py
data:
value:
apply_chat_template_kwargs:
enable_thinking: false
custom_cls:
name: null
path: null
datagen:
name: null
path: null
dataloader_num_workers: 8
filter_overlong_prompts: true
filter_overlong_prompts_workers: 1
image_key: images
image_patch_size: 14
max_prompt_length: 2048
max_response_length: 8192
prompt_key: prompt
return_full_prompt: false
return_multi_modal_inputs: true
return_raw_chat: true
return_raw_input_ids: false
reward_fn_key: data_source
sampler:
class_name: null
class_path: null
seed: null
shuffle: true
tokenizer: null
tool_config_path: null
train_batch_size: 32
train_files:
- /mnt/mole/SDPO/L2T/datasets/sciknoweval/physics/train.parquet
train_max_samples: 3200
truncation: error
trust_remote_code: true
use_shm: false
val_batch_size: null
val_files:
- /mnt/mole/SDPO/L2T/datasets/sciknoweval/physics/test.parquet
val_max_samples: -1
validation_shuffle: false
video_key: videos
global_profiler:
value:
_target_: verl.utils.profiler.ProfilerConfig
global_tool_config:
nsys:
_target_: verl.utils.profiler.config.NsightToolConfig
controller_nsight_options:
cuda-graph-trace: graph
cuda-memory-usage: "true"
trace: cuda,nvtx,cublas,ucx
discrete: false
worker_nsight_options:
capture-range: cudaProfilerApi
capture-range-end: null
cuda-graph-trace: graph
cuda-memory-usage: "true"
kill: none
trace: cuda,nvtx,cublas,ucx
torch_memory:
context: all
stack_depth: 32
stacks: all
trace_alloc_max_entries: 100000
profile_continuous_steps: false
save_path: outputs/profile
steps: null
tool: null
max_model_len:
value: 10240
ray_kwargs:
value:
ray_init:
_temp_dir: /tmp/ray_q3g_physics_grpo_Qwen3_8B
include_dashboard: false
num_cpus: null
runtime_env:
env_vars:
EXPERIMENT: qwen3gen-physics-GRPO-Qwen-Qwen3-8B-mbs8-train32-rollout8-lr1e-6-vllm0.8
TASK: datasets/sciknoweval/physics
USER: root
timeline_json_file: null
reward_manager:
value:
_target_: verl.trainer.config.config.RewardManagerConfig
module:
_target_: verl.trainer.config.config.ModuleConfig
name: custom_reward_manager
path: null
name: naive
source: register
reward_model:
value:
enable: false
enable_resource_pool: false
forward_max_token_len_per_gpu: 32768
launch_reward_fn_async: false
max_length: null
micro_batch_size: null
micro_batch_size_per_gpu: null
model:
external_lib: null
fsdp_config:
_target_: verl.workers.config.FSDPEngineConfig
forward_prefetch: false
fsdp_size: -1
param_offload: false
reshard_after_forward: true
wrap_policy:
min_num_params: 0
input_tokenizer: Qwen/Qwen3-8B
path: ~/models/FsfairX-LLaMA3-RM-v0.1
trust_remote_code: false
use_fused_kernels: false
use_remove_padding: false
use_shm: false
n_gpus_per_node: 8
nnodes: 0
num_workers: 1
profiler:
_target_: verl.utils.profiler.ProfilerConfig
all_ranks: false
enable: false
ranks: []
save_path: outputs/profile
tool: null
tool_config:
npu:
_target_: verl.utils.profiler.config.NPUToolConfig
analysis: true
contents: []
discrete: false
level: level0
nsys:
_target_: verl.utils.profiler.config.NsightToolConfig
discrete: false
torch:
_target_: verl.utils.profiler.config.TorchProfilerToolConfig
step_end: null
step_start: 0
torch_memory:
_target_: verl.utils.profiler.config.TorchMemoryToolConfig
stack_depth: 32
trace_alloc_max_entries: 100000
reward_loop_class_name: null
reward_loop_module_path: null
reward_loop_source: register
reward_manager: naive
rollout:
_target_: verl.workers.config.RolloutConfig
cudagraph_capture_sizes: null
data_parallel_size: 1
disable_log_stats: true
dtype: bfloat16
enable_chunked_prefill: true
enable_prefix_caching: true
enforce_eager: true
expert_parallel_size: 1
free_cache_engine: true
gpu_memory_utilization: 0.5
limit_images: null
load_format: auto
max_model_len: null
max_num_batched_tokens: 8192
max_num_seqs: 1024
name: ???
prompt_length: 2048
response_length: 2048
skip_tokenizer_init: false
tensor_model_parallel_size: 2
sandbox_fusion:
max_concurrent: 64
memory_limit_mb: 1024
url: null
strategy: fsdp
ulysses_sequence_parallel_size: 1
use_dynamic_bsz: false
use_reward_loop: false
trainer:
value:
balance_batch: true
critic_warmup: 0
default_hdfs_dir: null
default_local_dir: /mnt/mole/SDPO/L2T/checkpoints/datasets/sciknoweval/physics/qwen3gen-physics-GRPO-Qwen-Qwen3-8B-mbs8-train32-rollout8-lr1e-6-vllm0.8
del_local_ckpt_after_load: false
device: cuda
esi_redundant_time: 0
experiment_name: qwen3gen-physics-GRPO-Qwen-Qwen3-8B-mbs8-train32-rollout8-lr1e-6-vllm0.8
group_name: QWEN3-GRPO-generalization
log_val_generations: 0
logger:
- console
- wandb
max_actor_ckpt_to_keep: 100
max_critic_ckpt_to_keep: null
n_gpus_per_node: 8
nnodes: 1
project_name: SDPO-root
ray_wait_register_center_timeout: 300
resume_from_path: null
resume_mode: auto
rollout_data_dir: null
save_freq: 10
test_freq: 10
total_epochs: 30
total_training_steps: 100
use_legacy_worker_impl: auto
val_before_train: false
val_only: false
validation_data_dir: null
transfer_queue:
value:
enable: false
vars:
value:
ckpt_dir: /capstor/scratch/cscs/root/ttrl_runs/datasets/sciknoweval/physics
dir: /users/root/SDPO
log_dir: /users/root/output
task: datasets/sciknoweval/physics