_wandb: value: cli_version: 0.23.1 e: 1n1ucg63toheyodgdd2dcwkutcbcxhn8: args: - --node-ip-address=198.19.44.212 - --node-manager-port=36871 - --object-store-name=/tmp/ray_q3g_physics_grpo_Qwen3_8B/session_2026-07-03_03-18-52_070071_2279047/sockets/plasma_store - --raylet-name=/tmp/ray_q3g_physics_grpo_Qwen3_8B/session_2026-07-03_03-18-52_070071_2279047/sockets/raylet - --redis-address=None - --metrics-agent-port=56405 - --logging-rotate-bytes=536870912 - --logging-rotate-backup-count=5 - --runtime-env-agent-port=64277 - --gcs-address=198.19.44.212:64574 - --session-name=session_2026-07-03_03-18-52_070071_2279047 - --temp-dir=/tmp/ray_q3g_physics_grpo_Qwen3_8B - --webui= - --cluster-id=d7576956015edcd024cf2ca706acce8be3fb6519bb7b0c87bfccae89 - --startup-token=32 - --worker-launch-time-ms=1783048735496 - --node-id=e88e44745be3f8c4449808648428760387c2aa53671532e80c9f13b3 - --runtime-env-hash=1652680156 codePath: .venv/lib/python3.12/site-packages/ray/_private/workers/default_worker.py codePathLocal: .venv/lib/python3.12/site-packages/ray/_private/workers/default_worker.py cpu_count: 64 cpu_count_logical: 128 cudaVersion: "13.0" disk: /: total: "46086056050688" used: "25787742543872" email: jungsr1116@cau.ac.kr executable: /mnt/mole/SDPO/L2T/.venv/bin/python git: commit: ee6a667700697069b93db636aa937f970c1fb57a remote: https://github.com/jungseongryong/L2T.git gpu: NVIDIA H200 gpu_count: 8 gpu_nvidia: - architecture: Hopper cudaCores: 16896 memoryTotal: "150754820096" name: NVIDIA H200 uuid: GPU-c0451621-891c-7169-976d-71d81958db34 - architecture: Hopper cudaCores: 16896 memoryTotal: "150754820096" name: NVIDIA H200 uuid: GPU-e2a29dfa-7dad-6cd7-fb05-79d0060f27f0 - architecture: Hopper cudaCores: 16896 memoryTotal: "150754820096" name: NVIDIA H200 uuid: GPU-d69bb741-f2a7-a375-7f56-5c0d09d797b2 - architecture: Hopper cudaCores: 16896 memoryTotal: "150754820096" name: NVIDIA H200 uuid: GPU-1bbbea1e-7541-3126-1bfc-51b43a67577e - architecture: Hopper cudaCores: 16896 memoryTotal: "150754820096" name: NVIDIA H200 uuid: GPU-6e0055af-452d-3f46-ce21-0c71e347a608 - architecture: Hopper cudaCores: 16896 memoryTotal: "150754820096" name: NVIDIA H200 uuid: GPU-317fdb5b-b280-71d8-398e-a843ad897437 - architecture: Hopper cudaCores: 16896 memoryTotal: "150754820096" name: NVIDIA H200 uuid: GPU-f19d8161-d2dd-49a1-dcb8-5c341b5e4514 - architecture: Hopper cudaCores: 16896 memoryTotal: "150754820096" name: NVIDIA H200 uuid: GPU-961f5ebf-4bf8-6855-828b-c065af57b87c host: mole-workspace-rw memory: total: "2163980390400" os: Linux-6.8.0-71-generic-x86_64-with-glibc2.36 program: /mnt/mole/SDPO/L2T/.venv/lib/python3.12/site-packages/ray/_private/workers/default_worker.py python: CPython 3.12.13 root: /mnt/mole/SDPO/L2T startedAt: "2026-07-03T03:20:40.327312Z" writerId: 1n1ucg63toheyodgdd2dcwkutcbcxhn8 m: [] python_version: 3.12.13 t: "1": - 1 - 11 - 30 - 41 - 49 - 50 - 51 - 71 - 95 - 98 - 105 "2": - 1 - 11 - 30 - 41 - 49 - 50 - 51 - 71 - 95 - 98 - 105 "3": - 2 - 13 - 16 - 61 "4": 3.12.13 "5": 0.23.1 "6": 4.57.1 "12": 0.23.1 "13": linux-x86_64 actor_rollout_ref: value: actor: _target_: verl.workers.config.FSDPActorConfig calculate_entropy: false calculate_sum_pi_squared: false checkpoint: _target_: verl.trainer.config.CheckpointConfig async_save: false load_contents: - model - optimizer - extra save_contents: - model - optimizer - extra clip_ratio: 0.2 clip_ratio_c: 3 clip_ratio_high: 0.28 clip_ratio_low: 0.2 data_loader_seed: 42 entropy_checkpointing: false entropy_coeff: 0 entropy_from_logits_with_chunking: false freeze_vision_tower: false fsdp_config: _target_: verl.workers.config.FSDPEngineConfig dtype: bfloat16 entropy_checkpointing: false entropy_from_logits_with_chunking: false forward_only: false forward_prefetch: false fsdp_size: -1 full_determinism: false model_dtype: fp32 offload_policy: false optimizer_offload: false param_offload: false reshard_after_forward: true seed: 42 strategy: fsdp ulysses_sequence_parallel_size: 1 use_orig_params: false use_torch_compile: true wrap_policy: min_num_params: 0 grad_clip: 1 kl_loss_coef: 0 kl_loss_type: low_var_kl loss_agg_mode: token-mean loss_scale_factor: null optim: _target_: verl.workers.config.FSDPOptimizerConfig betas: - 0.9 - 0.999 clip_grad: 1 lr: 1e-06 lr_scheduler_type: constant lr_warmup_steps: 10 lr_warmup_steps_ratio: 0 min_lr_ratio: 0 num_cycles: 0.5 optimizer: AdamW optimizer_impl: torch.optim override_optimizer_config: null total_training_steps: 100 warmup_style: null weight_decay: 0.01 policy_loss: _target_: verl.workers.config.PolicyLossConfig clip_cov_lb: 1 clip_cov_ratio: 0.0002 clip_cov_ub: 5 kl_cov_ratio: 0.0002 loss_mode: vanilla ppo_kl_coef: 0.1 ppo_epochs: 1 ppo_max_token_len_per_gpu: 10240 ppo_micro_batch_size: null ppo_micro_batch_size_per_gpu: 1 ppo_mini_batch_size: 8 profiler: _target_: verl.utils.profiler.ProfilerConfig all_ranks: false enable: false ranks: [] save_path: outputs/profile tool: null tool_config: npu: _target_: verl.utils.profiler.config.NPUToolConfig analysis: true contents: [] discrete: false level: level0 nsys: _target_: verl.utils.profiler.config.NsightToolConfig discrete: false torch: _target_: verl.utils.profiler.config.TorchProfilerToolConfig step_end: null step_start: 0 torch_memory: _target_: verl.utils.profiler.config.TorchMemoryToolConfig stack_depth: 32 trace_alloc_max_entries: 100000 rollout_n: 8 router_replay: _target_: verl.workers.config.RouterReplayConfig mode: disabled record_file: null replay_file: null self_distillation: _target_: verl.workers.config.SelfDistillationConfig alpha: 0.5 distillation_add_tail: true distillation_topk: 100 dont_reprompt_on_self_success: true environment_feedback_only_without_solution: true evolving_teacher: _target_: verl.workers.config.EvolvingTeacherConfig enable: false loss_weight: 0 mask: all feedback_template: |4- The following is feedback from your unsuccessful earlier attempt: {feedback_raw} full_logit_distillation: true include_environment_feedback: true is_clip: 2 max_reprompt_len: 22528 remove_thinking_from_demonstration: false reprompt_template: |- {prompt}{solution}{feedback} Correctly solve the original question. reprompt_truncation: right solution_template: |4- Correct solution: {successful_previous_attempt} srpo_dynamic_weighting: false srpo_dynamic_weighting_temperature: 1 success_reward_threshold: 0.5 teacher_regularization: ema teacher_update_rate: 0 token_reweight_decay_steps: null token_reweight_eps_w: 0.2 token_reweight_lambda: 0.5 shuffle: false strategy: fsdp sum_pi_squared_checkpointing: false tau_neg: 1.05 tau_pos: 1 ulysses_sequence_parallel_size: 1 use_dynamic_bsz: false use_fused_kernels: false use_kl_loss: false use_prefix_grouper: false use_remove_padding: true use_torch_compile: true hybrid_engine: true model: _target_: verl.workers.config.HFModelConfig custom_chat_template: null enable_activation_offload: false enable_gradient_checkpointing: true exclude_modules: null external_lib: null fused_kernel_options: impl_backend: torch hf_config_path: null lora_adapter_path: null lora_alpha: 16 lora_rank: 0 path: Qwen/Qwen3-8B target_modules: all-linear tiled_mlp: enabled: false num_shards: 4 tokenizer_path: null trust_remote_code: true use_fused_kernels: false use_liger: false use_remove_padding: true use_shm: false nccl_timeout: 600 ref: _target_: verl.workers.config.FSDPActorConfig entropy_checkpointing: false entropy_from_logits_with_chunking: false fsdp_config: _target_: verl.workers.config.FSDPEngineConfig dtype: bfloat16 entropy_checkpointing: false entropy_from_logits_with_chunking: false forward_only: true forward_prefetch: false fsdp_size: -1 full_determinism: false model_dtype: fp32 offload_policy: false optimizer_offload: false param_offload: false reshard_after_forward: true seed: 42 strategy: fsdp ulysses_sequence_parallel_size: 1 use_orig_params: false use_torch_compile: true wrap_policy: min_num_params: 0 log_prob_max_token_len_per_gpu: 10240 log_prob_micro_batch_size: null log_prob_micro_batch_size_per_gpu: 1 log_prob_use_dynamic_bsz: false profiler: _target_: verl.utils.profiler.ProfilerConfig all_ranks: false enable: false ranks: [] save_path: outputs/profile tool: null tool_config: npu: _target_: verl.utils.profiler.config.NPUToolConfig analysis: true contents: [] discrete: false level: level0 nsys: _target_: verl.utils.profiler.config.NsightToolConfig discrete: false torch: _target_: verl.utils.profiler.config.TorchProfilerToolConfig step_end: null step_start: 0 torch_memory: _target_: verl.utils.profiler.config.TorchMemoryToolConfig stack_depth: 32 trace_alloc_max_entries: 100000 rollout_n: 8 router_replay: _target_: verl.workers.config.RouterReplayConfig mode: disabled record_file: null replay_file: null strategy: fsdp ulysses_sequence_parallel_size: 1 use_torch_compile: true rollout: _target_: verl.workers.config.RolloutConfig agent: _target_: verl.workers.config.AgentLoopConfig agent_loop_config_path: null custom_async_server: _target_: verl.workers.config.CustomAsyncServerConfig name: null path: null default_agent_loop: single_turn_agent num_workers: 8 calculate_log_probs: true cudagraph_capture_sizes: null data_parallel_size: 1 disable_log_stats: true do_sample: true dtype: bfloat16 enable_chunked_prefill: true enable_prefix_caching: true enable_rollout_routing_replay: false enforce_eager: false expert_parallel_size: 1 free_cache_engine: true gpu_memory_utilization: 0.8 ignore_eos: false layered_summon: false load_format: dummy log_prob_max_token_len_per_gpu: 10240 log_prob_micro_batch_size: null log_prob_micro_batch_size_per_gpu: 1 log_prob_use_dynamic_bsz: false logprobs_mode: processed_logprobs max_model_len: 10240 max_num_batched_tokens: 10240 max_num_seqs: 1024 mode: async multi_stage_wake_up: false multi_turn: _target_: verl.workers.config.MultiTurnConfig enable: false format: hermes interaction_config_path: null max_assistant_turns: null max_parallel_calls: 1 max_tool_response_length: 256 max_user_turns: null num_repeat_rollouts: null tokenization_sanity_check_mode: strict tool_config_path: null tool_response_truncate_side: middle use_inference_chat_template: false "n": 8 name: vllm over_sample_rate: 0 pipeline_model_parallel_size: 1 profiler: _target_: verl.utils.profiler.ProfilerConfig all_ranks: false enable: false ranks: [] save_path: outputs/profile tool: null tool_config: npu: _target_: verl.utils.profiler.config.NPUToolConfig analysis: true contents: [] discrete: false level: level0 nsys: _target_: verl.utils.profiler.config.NsightToolConfig discrete: false torch: _target_: verl.utils.profiler.config.TorchProfilerToolConfig step_end: null step_start: 0 torch_memory: _target_: verl.utils.profiler.config.TorchMemoryToolConfig stack_depth: 32 trace_alloc_max_entries: 100000 prometheus: _target_: verl.workers.config.PrometheusConfig enable: false file: /tmp/ray/session_latest/metrics/prometheus/prometheus.yml port: 9090 served_model_name: Qwen/Qwen3-8B prompt_length: 2048 quantization: null quantization_config_file: null response_length: 8192 scheduling_policy: fcfs skip_dump_dir: /tmp/rollout_dump skip_rollout: false skip_tokenizer_init: true temperature: 1 tensor_model_parallel_size: 2 top_k: -1 top_p: 1 trace: _target_: verl.workers.config.TraceConfig backend: null max_samples_per_step_per_worker: null token2text: false update_weights_bucket_megabytes: 512 val_kwargs: _target_: verl.workers.config.SamplingConfig do_sample: true "n": 16 temperature: 0.6 top_k: -1 top_p: 0.95 algorithm: value: _target_: verl.trainer.config.AlgoConfig adv_estimator: grpo gamma: 1 kl_ctrl: _target_: verl.trainer.config.KLControlConfig horizon: 10000 kl_coef: 0.001 target_kl: 0.1 type: fixed kl_penalty: kl lam: 1 norm_adv_by_std_in_grpo: false pf_ppo: reweight_method: pow weight_pow: 2 rollout_correction: bypass_mode: false loss_type: ppo_clip rollout_is: token rollout_is_batch_normalize: false rollout_is_threshold: 2 rollout_rs: null rollout_rs_threshold: null use_kl_in_reward: false use_pf_ppo: false critic: value: _target_: verl.workers.config.FSDPCriticConfig checkpoint: _target_: verl.trainer.config.CheckpointConfig async_save: false load_contents: - model - optimizer - extra save_contents: - model - optimizer - extra cliprange_value: 0.5 data_loader_seed: 42 enable: null forward_max_token_len_per_gpu: 32768 forward_micro_batch_size: null forward_micro_batch_size_per_gpu: null grad_clip: 1 loss_agg_mode: token-mean model: _target_: verl.workers.config.FSDPCriticModelCfg enable_activation_offload: false enable_gradient_checkpointing: true external_lib: null fsdp_config: _target_: verl.workers.config.FSDPEngineConfig dtype: bfloat16 entropy_checkpointing: false entropy_from_logits_with_chunking: false forward_only: false forward_prefetch: false fsdp_size: -1 full_determinism: false model_dtype: fp32 offload_policy: false optimizer_offload: false param_offload: false reshard_after_forward: true seed: 42 strategy: fsdp ulysses_sequence_parallel_size: 1 use_orig_params: false use_torch_compile: true wrap_policy: min_num_params: 0 lora_alpha: 16 lora_rank: 0 path: Qwen/Qwen3-8B target_modules: all-linear tiled_mlp: enabled: false num_shards: 4 tokenizer_path: Qwen/Qwen3-8B trust_remote_code: true use_remove_padding: false use_shm: false optim: _target_: verl.workers.config.FSDPOptimizerConfig betas: - 0.9 - 0.999 clip_grad: 1 lr: 1e-05 lr_scheduler_type: constant lr_warmup_steps: -1 lr_warmup_steps_ratio: 0 min_lr_ratio: 0 num_cycles: 0.5 optimizer: AdamW optimizer_impl: torch.optim override_optimizer_config: null total_training_steps: 100 warmup_style: null weight_decay: 0.01 ppo_epochs: 1 ppo_max_token_len_per_gpu: 32768 ppo_micro_batch_size: null ppo_micro_batch_size_per_gpu: null ppo_mini_batch_size: 8 profiler: _target_: verl.utils.profiler.ProfilerConfig all_ranks: false enable: false ranks: [] save_path: outputs/profile tool: null tool_config: npu: _target_: verl.utils.profiler.config.NPUToolConfig analysis: true contents: [] discrete: false level: level0 nsys: _target_: verl.utils.profiler.config.NsightToolConfig discrete: false torch: _target_: verl.utils.profiler.config.TorchProfilerToolConfig step_end: null step_start: 0 torch_memory: _target_: verl.utils.profiler.config.TorchMemoryToolConfig stack_depth: 32 trace_alloc_max_entries: 100000 rollout_n: 8 shuffle: false strategy: fsdp ulysses_sequence_parallel_size: 1 use_dynamic_bsz: false custom_reward_function: value: name: compute_score path: /mnt/mole/SDPO/L2T/verl/utils/reward_score/feedback/__init__.py data: value: apply_chat_template_kwargs: enable_thinking: false custom_cls: name: null path: null datagen: name: null path: null dataloader_num_workers: 8 filter_overlong_prompts: true filter_overlong_prompts_workers: 1 image_key: images image_patch_size: 14 max_prompt_length: 2048 max_response_length: 8192 prompt_key: prompt return_full_prompt: false return_multi_modal_inputs: true return_raw_chat: true return_raw_input_ids: false reward_fn_key: data_source sampler: class_name: null class_path: null seed: null shuffle: true tokenizer: null tool_config_path: null train_batch_size: 32 train_files: - /mnt/mole/SDPO/L2T/datasets/sciknoweval/physics/train.parquet train_max_samples: 3200 truncation: error trust_remote_code: true use_shm: false val_batch_size: null val_files: - /mnt/mole/SDPO/L2T/datasets/sciknoweval/physics/test.parquet val_max_samples: -1 validation_shuffle: false video_key: videos global_profiler: value: _target_: verl.utils.profiler.ProfilerConfig global_tool_config: nsys: _target_: verl.utils.profiler.config.NsightToolConfig controller_nsight_options: cuda-graph-trace: graph cuda-memory-usage: "true" trace: cuda,nvtx,cublas,ucx discrete: false worker_nsight_options: capture-range: cudaProfilerApi capture-range-end: null cuda-graph-trace: graph cuda-memory-usage: "true" kill: none trace: cuda,nvtx,cublas,ucx torch_memory: context: all stack_depth: 32 stacks: all trace_alloc_max_entries: 100000 profile_continuous_steps: false save_path: outputs/profile steps: null tool: null max_model_len: value: 10240 ray_kwargs: value: ray_init: _temp_dir: /tmp/ray_q3g_physics_grpo_Qwen3_8B include_dashboard: false num_cpus: null runtime_env: env_vars: EXPERIMENT: qwen3gen-physics-GRPO-Qwen-Qwen3-8B-mbs8-train32-rollout8-lr1e-6-vllm0.8 TASK: datasets/sciknoweval/physics USER: root timeline_json_file: null reward_manager: value: _target_: verl.trainer.config.config.RewardManagerConfig module: _target_: verl.trainer.config.config.ModuleConfig name: custom_reward_manager path: null name: naive source: register reward_model: value: enable: false enable_resource_pool: false forward_max_token_len_per_gpu: 32768 launch_reward_fn_async: false max_length: null micro_batch_size: null micro_batch_size_per_gpu: null model: external_lib: null fsdp_config: _target_: verl.workers.config.FSDPEngineConfig forward_prefetch: false fsdp_size: -1 param_offload: false reshard_after_forward: true wrap_policy: min_num_params: 0 input_tokenizer: Qwen/Qwen3-8B path: ~/models/FsfairX-LLaMA3-RM-v0.1 trust_remote_code: false use_fused_kernels: false use_remove_padding: false use_shm: false n_gpus_per_node: 8 nnodes: 0 num_workers: 1 profiler: _target_: verl.utils.profiler.ProfilerConfig all_ranks: false enable: false ranks: [] save_path: outputs/profile tool: null tool_config: npu: _target_: verl.utils.profiler.config.NPUToolConfig analysis: true contents: [] discrete: false level: level0 nsys: _target_: verl.utils.profiler.config.NsightToolConfig discrete: false torch: _target_: verl.utils.profiler.config.TorchProfilerToolConfig step_end: null step_start: 0 torch_memory: _target_: verl.utils.profiler.config.TorchMemoryToolConfig stack_depth: 32 trace_alloc_max_entries: 100000 reward_loop_class_name: null reward_loop_module_path: null reward_loop_source: register reward_manager: naive rollout: _target_: verl.workers.config.RolloutConfig cudagraph_capture_sizes: null data_parallel_size: 1 disable_log_stats: true dtype: bfloat16 enable_chunked_prefill: true enable_prefix_caching: true enforce_eager: true expert_parallel_size: 1 free_cache_engine: true gpu_memory_utilization: 0.5 limit_images: null load_format: auto max_model_len: null max_num_batched_tokens: 8192 max_num_seqs: 1024 name: ??? prompt_length: 2048 response_length: 2048 skip_tokenizer_init: false tensor_model_parallel_size: 2 sandbox_fusion: max_concurrent: 64 memory_limit_mb: 1024 url: null strategy: fsdp ulysses_sequence_parallel_size: 1 use_dynamic_bsz: false use_reward_loop: false trainer: value: balance_batch: true critic_warmup: 0 default_hdfs_dir: null default_local_dir: /mnt/mole/SDPO/L2T/checkpoints/datasets/sciknoweval/physics/qwen3gen-physics-GRPO-Qwen-Qwen3-8B-mbs8-train32-rollout8-lr1e-6-vllm0.8 del_local_ckpt_after_load: false device: cuda esi_redundant_time: 0 experiment_name: qwen3gen-physics-GRPO-Qwen-Qwen3-8B-mbs8-train32-rollout8-lr1e-6-vllm0.8 group_name: QWEN3-GRPO-generalization log_val_generations: 0 logger: - console - wandb max_actor_ckpt_to_keep: 100 max_critic_ckpt_to_keep: null n_gpus_per_node: 8 nnodes: 1 project_name: SDPO-root ray_wait_register_center_timeout: 300 resume_from_path: null resume_mode: auto rollout_data_dir: null save_freq: 10 test_freq: 10 total_epochs: 30 total_training_steps: 100 use_legacy_worker_impl: auto val_before_train: false val_only: false validation_data_dir: null transfer_queue: value: enable: false vars: value: ckpt_dir: /capstor/scratch/cscs/root/ttrl_runs/datasets/sciknoweval/physics dir: /users/root/SDPO log_dir: /users/root/output task: datasets/sciknoweval/physics