初始化项目,由ModelHub XC社区提供模型
Model: SeongryongJung/Qwen3-8B-Physics-GRPO-TR Source: Original Platform
This commit is contained in:
855
artifacts/config.yaml
Normal file
855
artifacts/config.yaml
Normal file
@@ -0,0 +1,855 @@
|
||||
_wandb:
|
||||
value:
|
||||
cli_version: 0.23.1
|
||||
e:
|
||||
1n1ucg63toheyodgdd2dcwkutcbcxhn8:
|
||||
args:
|
||||
- --node-ip-address=198.19.44.212
|
||||
- --node-manager-port=36871
|
||||
- --object-store-name=/tmp/ray_q3g_physics_grpo_Qwen3_8B/session_2026-07-03_03-18-52_070071_2279047/sockets/plasma_store
|
||||
- --raylet-name=/tmp/ray_q3g_physics_grpo_Qwen3_8B/session_2026-07-03_03-18-52_070071_2279047/sockets/raylet
|
||||
- --redis-address=None
|
||||
- --metrics-agent-port=56405
|
||||
- --logging-rotate-bytes=536870912
|
||||
- --logging-rotate-backup-count=5
|
||||
- --runtime-env-agent-port=64277
|
||||
- --gcs-address=198.19.44.212:64574
|
||||
- --session-name=session_2026-07-03_03-18-52_070071_2279047
|
||||
- --temp-dir=/tmp/ray_q3g_physics_grpo_Qwen3_8B
|
||||
- --webui=
|
||||
- --cluster-id=d7576956015edcd024cf2ca706acce8be3fb6519bb7b0c87bfccae89
|
||||
- --startup-token=32
|
||||
- --worker-launch-time-ms=1783048735496
|
||||
- --node-id=e88e44745be3f8c4449808648428760387c2aa53671532e80c9f13b3
|
||||
- --runtime-env-hash=1652680156
|
||||
codePath: .venv/lib/python3.12/site-packages/ray/_private/workers/default_worker.py
|
||||
codePathLocal: .venv/lib/python3.12/site-packages/ray/_private/workers/default_worker.py
|
||||
cpu_count: 64
|
||||
cpu_count_logical: 128
|
||||
cudaVersion: "13.0"
|
||||
disk:
|
||||
/:
|
||||
total: "46086056050688"
|
||||
used: "25787742543872"
|
||||
email: jungsr1116@cau.ac.kr
|
||||
executable: /mnt/mole/SDPO/L2T/.venv/bin/python
|
||||
git:
|
||||
commit: ee6a667700697069b93db636aa937f970c1fb57a
|
||||
remote: https://github.com/jungseongryong/L2T.git
|
||||
gpu: NVIDIA H200
|
||||
gpu_count: 8
|
||||
gpu_nvidia:
|
||||
- architecture: Hopper
|
||||
cudaCores: 16896
|
||||
memoryTotal: "150754820096"
|
||||
name: NVIDIA H200
|
||||
uuid: GPU-c0451621-891c-7169-976d-71d81958db34
|
||||
- architecture: Hopper
|
||||
cudaCores: 16896
|
||||
memoryTotal: "150754820096"
|
||||
name: NVIDIA H200
|
||||
uuid: GPU-e2a29dfa-7dad-6cd7-fb05-79d0060f27f0
|
||||
- architecture: Hopper
|
||||
cudaCores: 16896
|
||||
memoryTotal: "150754820096"
|
||||
name: NVIDIA H200
|
||||
uuid: GPU-d69bb741-f2a7-a375-7f56-5c0d09d797b2
|
||||
- architecture: Hopper
|
||||
cudaCores: 16896
|
||||
memoryTotal: "150754820096"
|
||||
name: NVIDIA H200
|
||||
uuid: GPU-1bbbea1e-7541-3126-1bfc-51b43a67577e
|
||||
- architecture: Hopper
|
||||
cudaCores: 16896
|
||||
memoryTotal: "150754820096"
|
||||
name: NVIDIA H200
|
||||
uuid: GPU-6e0055af-452d-3f46-ce21-0c71e347a608
|
||||
- architecture: Hopper
|
||||
cudaCores: 16896
|
||||
memoryTotal: "150754820096"
|
||||
name: NVIDIA H200
|
||||
uuid: GPU-317fdb5b-b280-71d8-398e-a843ad897437
|
||||
- architecture: Hopper
|
||||
cudaCores: 16896
|
||||
memoryTotal: "150754820096"
|
||||
name: NVIDIA H200
|
||||
uuid: GPU-f19d8161-d2dd-49a1-dcb8-5c341b5e4514
|
||||
- architecture: Hopper
|
||||
cudaCores: 16896
|
||||
memoryTotal: "150754820096"
|
||||
name: NVIDIA H200
|
||||
uuid: GPU-961f5ebf-4bf8-6855-828b-c065af57b87c
|
||||
host: mole-workspace-rw
|
||||
memory:
|
||||
total: "2163980390400"
|
||||
os: Linux-6.8.0-71-generic-x86_64-with-glibc2.36
|
||||
program: /mnt/mole/SDPO/L2T/.venv/lib/python3.12/site-packages/ray/_private/workers/default_worker.py
|
||||
python: CPython 3.12.13
|
||||
root: /mnt/mole/SDPO/L2T
|
||||
startedAt: "2026-07-03T03:20:40.327312Z"
|
||||
writerId: 1n1ucg63toheyodgdd2dcwkutcbcxhn8
|
||||
m: []
|
||||
python_version: 3.12.13
|
||||
t:
|
||||
"1":
|
||||
- 1
|
||||
- 11
|
||||
- 30
|
||||
- 41
|
||||
- 49
|
||||
- 50
|
||||
- 51
|
||||
- 71
|
||||
- 95
|
||||
- 98
|
||||
- 105
|
||||
"2":
|
||||
- 1
|
||||
- 11
|
||||
- 30
|
||||
- 41
|
||||
- 49
|
||||
- 50
|
||||
- 51
|
||||
- 71
|
||||
- 95
|
||||
- 98
|
||||
- 105
|
||||
"3":
|
||||
- 2
|
||||
- 13
|
||||
- 16
|
||||
- 61
|
||||
"4": 3.12.13
|
||||
"5": 0.23.1
|
||||
"6": 4.57.1
|
||||
"12": 0.23.1
|
||||
"13": linux-x86_64
|
||||
actor_rollout_ref:
|
||||
value:
|
||||
actor:
|
||||
_target_: verl.workers.config.FSDPActorConfig
|
||||
calculate_entropy: false
|
||||
calculate_sum_pi_squared: false
|
||||
checkpoint:
|
||||
_target_: verl.trainer.config.CheckpointConfig
|
||||
async_save: false
|
||||
load_contents:
|
||||
- model
|
||||
- optimizer
|
||||
- extra
|
||||
save_contents:
|
||||
- model
|
||||
- optimizer
|
||||
- extra
|
||||
clip_ratio: 0.2
|
||||
clip_ratio_c: 3
|
||||
clip_ratio_high: 0.28
|
||||
clip_ratio_low: 0.2
|
||||
data_loader_seed: 42
|
||||
entropy_checkpointing: false
|
||||
entropy_coeff: 0
|
||||
entropy_from_logits_with_chunking: false
|
||||
freeze_vision_tower: false
|
||||
fsdp_config:
|
||||
_target_: verl.workers.config.FSDPEngineConfig
|
||||
dtype: bfloat16
|
||||
entropy_checkpointing: false
|
||||
entropy_from_logits_with_chunking: false
|
||||
forward_only: false
|
||||
forward_prefetch: false
|
||||
fsdp_size: -1
|
||||
full_determinism: false
|
||||
model_dtype: fp32
|
||||
offload_policy: false
|
||||
optimizer_offload: false
|
||||
param_offload: false
|
||||
reshard_after_forward: true
|
||||
seed: 42
|
||||
strategy: fsdp
|
||||
ulysses_sequence_parallel_size: 1
|
||||
use_orig_params: false
|
||||
use_torch_compile: true
|
||||
wrap_policy:
|
||||
min_num_params: 0
|
||||
grad_clip: 1
|
||||
kl_loss_coef: 0
|
||||
kl_loss_type: low_var_kl
|
||||
loss_agg_mode: token-mean
|
||||
loss_scale_factor: null
|
||||
optim:
|
||||
_target_: verl.workers.config.FSDPOptimizerConfig
|
||||
betas:
|
||||
- 0.9
|
||||
- 0.999
|
||||
clip_grad: 1
|
||||
lr: 1e-06
|
||||
lr_scheduler_type: constant
|
||||
lr_warmup_steps: 10
|
||||
lr_warmup_steps_ratio: 0
|
||||
min_lr_ratio: 0
|
||||
num_cycles: 0.5
|
||||
optimizer: AdamW
|
||||
optimizer_impl: torch.optim
|
||||
override_optimizer_config: null
|
||||
total_training_steps: 100
|
||||
warmup_style: null
|
||||
weight_decay: 0.01
|
||||
policy_loss:
|
||||
_target_: verl.workers.config.PolicyLossConfig
|
||||
clip_cov_lb: 1
|
||||
clip_cov_ratio: 0.0002
|
||||
clip_cov_ub: 5
|
||||
kl_cov_ratio: 0.0002
|
||||
loss_mode: vanilla
|
||||
ppo_kl_coef: 0.1
|
||||
ppo_epochs: 1
|
||||
ppo_max_token_len_per_gpu: 10240
|
||||
ppo_micro_batch_size: null
|
||||
ppo_micro_batch_size_per_gpu: 1
|
||||
ppo_mini_batch_size: 8
|
||||
profiler:
|
||||
_target_: verl.utils.profiler.ProfilerConfig
|
||||
all_ranks: false
|
||||
enable: false
|
||||
ranks: []
|
||||
save_path: outputs/profile
|
||||
tool: null
|
||||
tool_config:
|
||||
npu:
|
||||
_target_: verl.utils.profiler.config.NPUToolConfig
|
||||
analysis: true
|
||||
contents: []
|
||||
discrete: false
|
||||
level: level0
|
||||
nsys:
|
||||
_target_: verl.utils.profiler.config.NsightToolConfig
|
||||
discrete: false
|
||||
torch:
|
||||
_target_: verl.utils.profiler.config.TorchProfilerToolConfig
|
||||
step_end: null
|
||||
step_start: 0
|
||||
torch_memory:
|
||||
_target_: verl.utils.profiler.config.TorchMemoryToolConfig
|
||||
stack_depth: 32
|
||||
trace_alloc_max_entries: 100000
|
||||
rollout_n: 8
|
||||
router_replay:
|
||||
_target_: verl.workers.config.RouterReplayConfig
|
||||
mode: disabled
|
||||
record_file: null
|
||||
replay_file: null
|
||||
self_distillation:
|
||||
_target_: verl.workers.config.SelfDistillationConfig
|
||||
alpha: 0.5
|
||||
distillation_add_tail: true
|
||||
distillation_topk: 100
|
||||
dont_reprompt_on_self_success: true
|
||||
environment_feedback_only_without_solution: true
|
||||
evolving_teacher:
|
||||
_target_: verl.workers.config.EvolvingTeacherConfig
|
||||
enable: false
|
||||
loss_weight: 0
|
||||
mask: all
|
||||
feedback_template: |4-
|
||||
The following is feedback from your unsuccessful earlier attempt:
|
||||
|
||||
{feedback_raw}
|
||||
full_logit_distillation: true
|
||||
include_environment_feedback: true
|
||||
is_clip: 2
|
||||
max_reprompt_len: 22528
|
||||
remove_thinking_from_demonstration: false
|
||||
reprompt_template: |-
|
||||
{prompt}{solution}{feedback}
|
||||
|
||||
Correctly solve the original question.
|
||||
reprompt_truncation: right
|
||||
solution_template: |4-
|
||||
Correct solution:
|
||||
|
||||
{successful_previous_attempt}
|
||||
srpo_dynamic_weighting: false
|
||||
srpo_dynamic_weighting_temperature: 1
|
||||
success_reward_threshold: 0.5
|
||||
teacher_regularization: ema
|
||||
teacher_update_rate: 0
|
||||
token_reweight_decay_steps: null
|
||||
token_reweight_eps_w: 0.2
|
||||
token_reweight_lambda: 0.5
|
||||
shuffle: false
|
||||
strategy: fsdp
|
||||
sum_pi_squared_checkpointing: false
|
||||
tau_neg: 1.05
|
||||
tau_pos: 1
|
||||
ulysses_sequence_parallel_size: 1
|
||||
use_dynamic_bsz: false
|
||||
use_fused_kernels: false
|
||||
use_kl_loss: false
|
||||
use_prefix_grouper: false
|
||||
use_remove_padding: true
|
||||
use_torch_compile: true
|
||||
hybrid_engine: true
|
||||
model:
|
||||
_target_: verl.workers.config.HFModelConfig
|
||||
custom_chat_template: null
|
||||
enable_activation_offload: false
|
||||
enable_gradient_checkpointing: true
|
||||
exclude_modules: null
|
||||
external_lib: null
|
||||
fused_kernel_options:
|
||||
impl_backend: torch
|
||||
hf_config_path: null
|
||||
lora_adapter_path: null
|
||||
lora_alpha: 16
|
||||
lora_rank: 0
|
||||
path: Qwen/Qwen3-8B
|
||||
target_modules: all-linear
|
||||
tiled_mlp:
|
||||
enabled: false
|
||||
num_shards: 4
|
||||
tokenizer_path: null
|
||||
trust_remote_code: true
|
||||
use_fused_kernels: false
|
||||
use_liger: false
|
||||
use_remove_padding: true
|
||||
use_shm: false
|
||||
nccl_timeout: 600
|
||||
ref:
|
||||
_target_: verl.workers.config.FSDPActorConfig
|
||||
entropy_checkpointing: false
|
||||
entropy_from_logits_with_chunking: false
|
||||
fsdp_config:
|
||||
_target_: verl.workers.config.FSDPEngineConfig
|
||||
dtype: bfloat16
|
||||
entropy_checkpointing: false
|
||||
entropy_from_logits_with_chunking: false
|
||||
forward_only: true
|
||||
forward_prefetch: false
|
||||
fsdp_size: -1
|
||||
full_determinism: false
|
||||
model_dtype: fp32
|
||||
offload_policy: false
|
||||
optimizer_offload: false
|
||||
param_offload: false
|
||||
reshard_after_forward: true
|
||||
seed: 42
|
||||
strategy: fsdp
|
||||
ulysses_sequence_parallel_size: 1
|
||||
use_orig_params: false
|
||||
use_torch_compile: true
|
||||
wrap_policy:
|
||||
min_num_params: 0
|
||||
log_prob_max_token_len_per_gpu: 10240
|
||||
log_prob_micro_batch_size: null
|
||||
log_prob_micro_batch_size_per_gpu: 1
|
||||
log_prob_use_dynamic_bsz: false
|
||||
profiler:
|
||||
_target_: verl.utils.profiler.ProfilerConfig
|
||||
all_ranks: false
|
||||
enable: false
|
||||
ranks: []
|
||||
save_path: outputs/profile
|
||||
tool: null
|
||||
tool_config:
|
||||
npu:
|
||||
_target_: verl.utils.profiler.config.NPUToolConfig
|
||||
analysis: true
|
||||
contents: []
|
||||
discrete: false
|
||||
level: level0
|
||||
nsys:
|
||||
_target_: verl.utils.profiler.config.NsightToolConfig
|
||||
discrete: false
|
||||
torch:
|
||||
_target_: verl.utils.profiler.config.TorchProfilerToolConfig
|
||||
step_end: null
|
||||
step_start: 0
|
||||
torch_memory:
|
||||
_target_: verl.utils.profiler.config.TorchMemoryToolConfig
|
||||
stack_depth: 32
|
||||
trace_alloc_max_entries: 100000
|
||||
rollout_n: 8
|
||||
router_replay:
|
||||
_target_: verl.workers.config.RouterReplayConfig
|
||||
mode: disabled
|
||||
record_file: null
|
||||
replay_file: null
|
||||
strategy: fsdp
|
||||
ulysses_sequence_parallel_size: 1
|
||||
use_torch_compile: true
|
||||
rollout:
|
||||
_target_: verl.workers.config.RolloutConfig
|
||||
agent:
|
||||
_target_: verl.workers.config.AgentLoopConfig
|
||||
agent_loop_config_path: null
|
||||
custom_async_server:
|
||||
_target_: verl.workers.config.CustomAsyncServerConfig
|
||||
name: null
|
||||
path: null
|
||||
default_agent_loop: single_turn_agent
|
||||
num_workers: 8
|
||||
calculate_log_probs: true
|
||||
cudagraph_capture_sizes: null
|
||||
data_parallel_size: 1
|
||||
disable_log_stats: true
|
||||
do_sample: true
|
||||
dtype: bfloat16
|
||||
enable_chunked_prefill: true
|
||||
enable_prefix_caching: true
|
||||
enable_rollout_routing_replay: false
|
||||
enforce_eager: false
|
||||
expert_parallel_size: 1
|
||||
free_cache_engine: true
|
||||
gpu_memory_utilization: 0.8
|
||||
ignore_eos: false
|
||||
layered_summon: false
|
||||
load_format: dummy
|
||||
log_prob_max_token_len_per_gpu: 10240
|
||||
log_prob_micro_batch_size: null
|
||||
log_prob_micro_batch_size_per_gpu: 1
|
||||
log_prob_use_dynamic_bsz: false
|
||||
logprobs_mode: processed_logprobs
|
||||
max_model_len: 10240
|
||||
max_num_batched_tokens: 10240
|
||||
max_num_seqs: 1024
|
||||
mode: async
|
||||
multi_stage_wake_up: false
|
||||
multi_turn:
|
||||
_target_: verl.workers.config.MultiTurnConfig
|
||||
enable: false
|
||||
format: hermes
|
||||
interaction_config_path: null
|
||||
max_assistant_turns: null
|
||||
max_parallel_calls: 1
|
||||
max_tool_response_length: 256
|
||||
max_user_turns: null
|
||||
num_repeat_rollouts: null
|
||||
tokenization_sanity_check_mode: strict
|
||||
tool_config_path: null
|
||||
tool_response_truncate_side: middle
|
||||
use_inference_chat_template: false
|
||||
"n": 8
|
||||
name: vllm
|
||||
over_sample_rate: 0
|
||||
pipeline_model_parallel_size: 1
|
||||
profiler:
|
||||
_target_: verl.utils.profiler.ProfilerConfig
|
||||
all_ranks: false
|
||||
enable: false
|
||||
ranks: []
|
||||
save_path: outputs/profile
|
||||
tool: null
|
||||
tool_config:
|
||||
npu:
|
||||
_target_: verl.utils.profiler.config.NPUToolConfig
|
||||
analysis: true
|
||||
contents: []
|
||||
discrete: false
|
||||
level: level0
|
||||
nsys:
|
||||
_target_: verl.utils.profiler.config.NsightToolConfig
|
||||
discrete: false
|
||||
torch:
|
||||
_target_: verl.utils.profiler.config.TorchProfilerToolConfig
|
||||
step_end: null
|
||||
step_start: 0
|
||||
torch_memory:
|
||||
_target_: verl.utils.profiler.config.TorchMemoryToolConfig
|
||||
stack_depth: 32
|
||||
trace_alloc_max_entries: 100000
|
||||
prometheus:
|
||||
_target_: verl.workers.config.PrometheusConfig
|
||||
enable: false
|
||||
file: /tmp/ray/session_latest/metrics/prometheus/prometheus.yml
|
||||
port: 9090
|
||||
served_model_name: Qwen/Qwen3-8B
|
||||
prompt_length: 2048
|
||||
quantization: null
|
||||
quantization_config_file: null
|
||||
response_length: 8192
|
||||
scheduling_policy: fcfs
|
||||
skip_dump_dir: /tmp/rollout_dump
|
||||
skip_rollout: false
|
||||
skip_tokenizer_init: true
|
||||
temperature: 1
|
||||
tensor_model_parallel_size: 2
|
||||
top_k: -1
|
||||
top_p: 1
|
||||
trace:
|
||||
_target_: verl.workers.config.TraceConfig
|
||||
backend: null
|
||||
max_samples_per_step_per_worker: null
|
||||
token2text: false
|
||||
update_weights_bucket_megabytes: 512
|
||||
val_kwargs:
|
||||
_target_: verl.workers.config.SamplingConfig
|
||||
do_sample: true
|
||||
"n": 16
|
||||
temperature: 0.6
|
||||
top_k: -1
|
||||
top_p: 0.95
|
||||
algorithm:
|
||||
value:
|
||||
_target_: verl.trainer.config.AlgoConfig
|
||||
adv_estimator: grpo
|
||||
gamma: 1
|
||||
kl_ctrl:
|
||||
_target_: verl.trainer.config.KLControlConfig
|
||||
horizon: 10000
|
||||
kl_coef: 0.001
|
||||
target_kl: 0.1
|
||||
type: fixed
|
||||
kl_penalty: kl
|
||||
lam: 1
|
||||
norm_adv_by_std_in_grpo: false
|
||||
pf_ppo:
|
||||
reweight_method: pow
|
||||
weight_pow: 2
|
||||
rollout_correction:
|
||||
bypass_mode: false
|
||||
loss_type: ppo_clip
|
||||
rollout_is: token
|
||||
rollout_is_batch_normalize: false
|
||||
rollout_is_threshold: 2
|
||||
rollout_rs: null
|
||||
rollout_rs_threshold: null
|
||||
use_kl_in_reward: false
|
||||
use_pf_ppo: false
|
||||
critic:
|
||||
value:
|
||||
_target_: verl.workers.config.FSDPCriticConfig
|
||||
checkpoint:
|
||||
_target_: verl.trainer.config.CheckpointConfig
|
||||
async_save: false
|
||||
load_contents:
|
||||
- model
|
||||
- optimizer
|
||||
- extra
|
||||
save_contents:
|
||||
- model
|
||||
- optimizer
|
||||
- extra
|
||||
cliprange_value: 0.5
|
||||
data_loader_seed: 42
|
||||
enable: null
|
||||
forward_max_token_len_per_gpu: 32768
|
||||
forward_micro_batch_size: null
|
||||
forward_micro_batch_size_per_gpu: null
|
||||
grad_clip: 1
|
||||
loss_agg_mode: token-mean
|
||||
model:
|
||||
_target_: verl.workers.config.FSDPCriticModelCfg
|
||||
enable_activation_offload: false
|
||||
enable_gradient_checkpointing: true
|
||||
external_lib: null
|
||||
fsdp_config:
|
||||
_target_: verl.workers.config.FSDPEngineConfig
|
||||
dtype: bfloat16
|
||||
entropy_checkpointing: false
|
||||
entropy_from_logits_with_chunking: false
|
||||
forward_only: false
|
||||
forward_prefetch: false
|
||||
fsdp_size: -1
|
||||
full_determinism: false
|
||||
model_dtype: fp32
|
||||
offload_policy: false
|
||||
optimizer_offload: false
|
||||
param_offload: false
|
||||
reshard_after_forward: true
|
||||
seed: 42
|
||||
strategy: fsdp
|
||||
ulysses_sequence_parallel_size: 1
|
||||
use_orig_params: false
|
||||
use_torch_compile: true
|
||||
wrap_policy:
|
||||
min_num_params: 0
|
||||
lora_alpha: 16
|
||||
lora_rank: 0
|
||||
path: Qwen/Qwen3-8B
|
||||
target_modules: all-linear
|
||||
tiled_mlp:
|
||||
enabled: false
|
||||
num_shards: 4
|
||||
tokenizer_path: Qwen/Qwen3-8B
|
||||
trust_remote_code: true
|
||||
use_remove_padding: false
|
||||
use_shm: false
|
||||
optim:
|
||||
_target_: verl.workers.config.FSDPOptimizerConfig
|
||||
betas:
|
||||
- 0.9
|
||||
- 0.999
|
||||
clip_grad: 1
|
||||
lr: 1e-05
|
||||
lr_scheduler_type: constant
|
||||
lr_warmup_steps: -1
|
||||
lr_warmup_steps_ratio: 0
|
||||
min_lr_ratio: 0
|
||||
num_cycles: 0.5
|
||||
optimizer: AdamW
|
||||
optimizer_impl: torch.optim
|
||||
override_optimizer_config: null
|
||||
total_training_steps: 100
|
||||
warmup_style: null
|
||||
weight_decay: 0.01
|
||||
ppo_epochs: 1
|
||||
ppo_max_token_len_per_gpu: 32768
|
||||
ppo_micro_batch_size: null
|
||||
ppo_micro_batch_size_per_gpu: null
|
||||
ppo_mini_batch_size: 8
|
||||
profiler:
|
||||
_target_: verl.utils.profiler.ProfilerConfig
|
||||
all_ranks: false
|
||||
enable: false
|
||||
ranks: []
|
||||
save_path: outputs/profile
|
||||
tool: null
|
||||
tool_config:
|
||||
npu:
|
||||
_target_: verl.utils.profiler.config.NPUToolConfig
|
||||
analysis: true
|
||||
contents: []
|
||||
discrete: false
|
||||
level: level0
|
||||
nsys:
|
||||
_target_: verl.utils.profiler.config.NsightToolConfig
|
||||
discrete: false
|
||||
torch:
|
||||
_target_: verl.utils.profiler.config.TorchProfilerToolConfig
|
||||
step_end: null
|
||||
step_start: 0
|
||||
torch_memory:
|
||||
_target_: verl.utils.profiler.config.TorchMemoryToolConfig
|
||||
stack_depth: 32
|
||||
trace_alloc_max_entries: 100000
|
||||
rollout_n: 8
|
||||
shuffle: false
|
||||
strategy: fsdp
|
||||
ulysses_sequence_parallel_size: 1
|
||||
use_dynamic_bsz: false
|
||||
custom_reward_function:
|
||||
value:
|
||||
name: compute_score
|
||||
path: /mnt/mole/SDPO/L2T/verl/utils/reward_score/feedback/__init__.py
|
||||
data:
|
||||
value:
|
||||
apply_chat_template_kwargs:
|
||||
enable_thinking: false
|
||||
custom_cls:
|
||||
name: null
|
||||
path: null
|
||||
datagen:
|
||||
name: null
|
||||
path: null
|
||||
dataloader_num_workers: 8
|
||||
filter_overlong_prompts: true
|
||||
filter_overlong_prompts_workers: 1
|
||||
image_key: images
|
||||
image_patch_size: 14
|
||||
max_prompt_length: 2048
|
||||
max_response_length: 8192
|
||||
prompt_key: prompt
|
||||
return_full_prompt: false
|
||||
return_multi_modal_inputs: true
|
||||
return_raw_chat: true
|
||||
return_raw_input_ids: false
|
||||
reward_fn_key: data_source
|
||||
sampler:
|
||||
class_name: null
|
||||
class_path: null
|
||||
seed: null
|
||||
shuffle: true
|
||||
tokenizer: null
|
||||
tool_config_path: null
|
||||
train_batch_size: 32
|
||||
train_files:
|
||||
- /mnt/mole/SDPO/L2T/datasets/sciknoweval/physics/train.parquet
|
||||
train_max_samples: 3200
|
||||
truncation: error
|
||||
trust_remote_code: true
|
||||
use_shm: false
|
||||
val_batch_size: null
|
||||
val_files:
|
||||
- /mnt/mole/SDPO/L2T/datasets/sciknoweval/physics/test.parquet
|
||||
val_max_samples: -1
|
||||
validation_shuffle: false
|
||||
video_key: videos
|
||||
global_profiler:
|
||||
value:
|
||||
_target_: verl.utils.profiler.ProfilerConfig
|
||||
global_tool_config:
|
||||
nsys:
|
||||
_target_: verl.utils.profiler.config.NsightToolConfig
|
||||
controller_nsight_options:
|
||||
cuda-graph-trace: graph
|
||||
cuda-memory-usage: "true"
|
||||
trace: cuda,nvtx,cublas,ucx
|
||||
discrete: false
|
||||
worker_nsight_options:
|
||||
capture-range: cudaProfilerApi
|
||||
capture-range-end: null
|
||||
cuda-graph-trace: graph
|
||||
cuda-memory-usage: "true"
|
||||
kill: none
|
||||
trace: cuda,nvtx,cublas,ucx
|
||||
torch_memory:
|
||||
context: all
|
||||
stack_depth: 32
|
||||
stacks: all
|
||||
trace_alloc_max_entries: 100000
|
||||
profile_continuous_steps: false
|
||||
save_path: outputs/profile
|
||||
steps: null
|
||||
tool: null
|
||||
max_model_len:
|
||||
value: 10240
|
||||
ray_kwargs:
|
||||
value:
|
||||
ray_init:
|
||||
_temp_dir: /tmp/ray_q3g_physics_grpo_Qwen3_8B
|
||||
include_dashboard: false
|
||||
num_cpus: null
|
||||
runtime_env:
|
||||
env_vars:
|
||||
EXPERIMENT: qwen3gen-physics-GRPO-Qwen-Qwen3-8B-mbs8-train32-rollout8-lr1e-6-vllm0.8
|
||||
TASK: datasets/sciknoweval/physics
|
||||
USER: root
|
||||
timeline_json_file: null
|
||||
reward_manager:
|
||||
value:
|
||||
_target_: verl.trainer.config.config.RewardManagerConfig
|
||||
module:
|
||||
_target_: verl.trainer.config.config.ModuleConfig
|
||||
name: custom_reward_manager
|
||||
path: null
|
||||
name: naive
|
||||
source: register
|
||||
reward_model:
|
||||
value:
|
||||
enable: false
|
||||
enable_resource_pool: false
|
||||
forward_max_token_len_per_gpu: 32768
|
||||
launch_reward_fn_async: false
|
||||
max_length: null
|
||||
micro_batch_size: null
|
||||
micro_batch_size_per_gpu: null
|
||||
model:
|
||||
external_lib: null
|
||||
fsdp_config:
|
||||
_target_: verl.workers.config.FSDPEngineConfig
|
||||
forward_prefetch: false
|
||||
fsdp_size: -1
|
||||
param_offload: false
|
||||
reshard_after_forward: true
|
||||
wrap_policy:
|
||||
min_num_params: 0
|
||||
input_tokenizer: Qwen/Qwen3-8B
|
||||
path: ~/models/FsfairX-LLaMA3-RM-v0.1
|
||||
trust_remote_code: false
|
||||
use_fused_kernels: false
|
||||
use_remove_padding: false
|
||||
use_shm: false
|
||||
n_gpus_per_node: 8
|
||||
nnodes: 0
|
||||
num_workers: 1
|
||||
profiler:
|
||||
_target_: verl.utils.profiler.ProfilerConfig
|
||||
all_ranks: false
|
||||
enable: false
|
||||
ranks: []
|
||||
save_path: outputs/profile
|
||||
tool: null
|
||||
tool_config:
|
||||
npu:
|
||||
_target_: verl.utils.profiler.config.NPUToolConfig
|
||||
analysis: true
|
||||
contents: []
|
||||
discrete: false
|
||||
level: level0
|
||||
nsys:
|
||||
_target_: verl.utils.profiler.config.NsightToolConfig
|
||||
discrete: false
|
||||
torch:
|
||||
_target_: verl.utils.profiler.config.TorchProfilerToolConfig
|
||||
step_end: null
|
||||
step_start: 0
|
||||
torch_memory:
|
||||
_target_: verl.utils.profiler.config.TorchMemoryToolConfig
|
||||
stack_depth: 32
|
||||
trace_alloc_max_entries: 100000
|
||||
reward_loop_class_name: null
|
||||
reward_loop_module_path: null
|
||||
reward_loop_source: register
|
||||
reward_manager: naive
|
||||
rollout:
|
||||
_target_: verl.workers.config.RolloutConfig
|
||||
cudagraph_capture_sizes: null
|
||||
data_parallel_size: 1
|
||||
disable_log_stats: true
|
||||
dtype: bfloat16
|
||||
enable_chunked_prefill: true
|
||||
enable_prefix_caching: true
|
||||
enforce_eager: true
|
||||
expert_parallel_size: 1
|
||||
free_cache_engine: true
|
||||
gpu_memory_utilization: 0.5
|
||||
limit_images: null
|
||||
load_format: auto
|
||||
max_model_len: null
|
||||
max_num_batched_tokens: 8192
|
||||
max_num_seqs: 1024
|
||||
name: ???
|
||||
prompt_length: 2048
|
||||
response_length: 2048
|
||||
skip_tokenizer_init: false
|
||||
tensor_model_parallel_size: 2
|
||||
sandbox_fusion:
|
||||
max_concurrent: 64
|
||||
memory_limit_mb: 1024
|
||||
url: null
|
||||
strategy: fsdp
|
||||
ulysses_sequence_parallel_size: 1
|
||||
use_dynamic_bsz: false
|
||||
use_reward_loop: false
|
||||
trainer:
|
||||
value:
|
||||
balance_batch: true
|
||||
critic_warmup: 0
|
||||
default_hdfs_dir: null
|
||||
default_local_dir: /mnt/mole/SDPO/L2T/checkpoints/datasets/sciknoweval/physics/qwen3gen-physics-GRPO-Qwen-Qwen3-8B-mbs8-train32-rollout8-lr1e-6-vllm0.8
|
||||
del_local_ckpt_after_load: false
|
||||
device: cuda
|
||||
esi_redundant_time: 0
|
||||
experiment_name: qwen3gen-physics-GRPO-Qwen-Qwen3-8B-mbs8-train32-rollout8-lr1e-6-vllm0.8
|
||||
group_name: QWEN3-GRPO-generalization
|
||||
log_val_generations: 0
|
||||
logger:
|
||||
- console
|
||||
- wandb
|
||||
max_actor_ckpt_to_keep: 100
|
||||
max_critic_ckpt_to_keep: null
|
||||
n_gpus_per_node: 8
|
||||
nnodes: 1
|
||||
project_name: SDPO-root
|
||||
ray_wait_register_center_timeout: 300
|
||||
resume_from_path: null
|
||||
resume_mode: auto
|
||||
rollout_data_dir: null
|
||||
save_freq: 10
|
||||
test_freq: 10
|
||||
total_epochs: 30
|
||||
total_training_steps: 100
|
||||
use_legacy_worker_impl: auto
|
||||
val_before_train: false
|
||||
val_only: false
|
||||
validation_data_dir: null
|
||||
transfer_queue:
|
||||
value:
|
||||
enable: false
|
||||
vars:
|
||||
value:
|
||||
ckpt_dir: /capstor/scratch/cscs/root/ttrl_runs/datasets/sciknoweval/physics
|
||||
dir: /users/root/SDPO
|
||||
log_dir: /users/root/output
|
||||
task: datasets/sciknoweval/physics
|
||||
1189
artifacts/output.log
Normal file
1189
artifacts/output.log
Normal file
File diff suppressed because it is too large
Load Diff
7823
artifacts/queue.log
Normal file
7823
artifacts/queue.log
Normal file
File diff suppressed because one or more lines are too long
286
artifacts/requirements.txt
Normal file
286
artifacts/requirements.txt
Normal file
@@ -0,0 +1,286 @@
|
||||
colorama==0.4.6
|
||||
psutil==7.1.3
|
||||
wheel==0.47.0
|
||||
pip==26.1.2
|
||||
mpmath==1.3.0
|
||||
typing_extensions==4.15.0
|
||||
pillow==12.2.0
|
||||
nvidia-cuda-runtime-cu12==12.8.90
|
||||
nvidia-cuda-nvrtc-cu12==12.8.93
|
||||
nvidia-cusparse-cu12==12.5.8.93
|
||||
nvidia-cufft-cu12==11.3.3.83
|
||||
nvidia-cuda-cupti-cu12==12.8.90
|
||||
nvidia-cublas-cu12==12.8.4.1
|
||||
llguidance==1.3.0
|
||||
anthropic==0.71.0
|
||||
nodeenv==1.10.0
|
||||
networkx==3.6.1
|
||||
MarkupSafe==3.0.3
|
||||
filelock==3.29.0
|
||||
nvidia-cudnn-cu12==9.10.2.21
|
||||
transformers==4.57.1
|
||||
Jinja2==3.1.6
|
||||
nvidia-cusolver-cu12==11.7.3.90
|
||||
torch==2.9.0
|
||||
rich-toolkit==0.20.1
|
||||
z3-solver==4.15.4.0
|
||||
word2number==1.1
|
||||
pytz==2026.2
|
||||
pylatexenc==2.10
|
||||
py-spy==0.4.2
|
||||
opencensus-context==0.1.3
|
||||
distlib==0.4.3
|
||||
colorful==0.5.8
|
||||
antlr4-python3-runtime==4.9.3
|
||||
zipp==4.1.0
|
||||
xxhash==3.7.0
|
||||
wrapt==2.2.1
|
||||
Werkzeug==3.1.8
|
||||
urllib3==2.7.0
|
||||
tzdata==2026.2
|
||||
typing-inspection==0.4.2
|
||||
tqdm==4.68.2
|
||||
tensorboard-data-server==0.7.2
|
||||
smmap==5.0.3
|
||||
six==1.17.0
|
||||
safetensors==0.8.0
|
||||
rpds-py==2026.5.1
|
||||
regex==2026.5.9
|
||||
pyzmq==27.1.0
|
||||
PyYAML==6.0.3
|
||||
Pygments==2.20.0
|
||||
pydantic_core==2.46.4
|
||||
pycparser==3.0
|
||||
pybind11==3.0.1
|
||||
pyasn1==0.6.3
|
||||
pyarrow==22.0.0
|
||||
psutil==7.1.3
|
||||
protobuf==6.33.6
|
||||
propcache==0.5.2
|
||||
prometheus_client==0.25.0
|
||||
pluggy==1.6.0
|
||||
platformdirs==4.10.0
|
||||
httpcore==1.0.9
|
||||
packaging==25.0
|
||||
orjson==3.11.9
|
||||
opentelemetry-api==1.42.1
|
||||
numpy==2.1.0
|
||||
multidict==6.7.1
|
||||
msgpack==1.2.0
|
||||
Markdown==3.10.2
|
||||
iniconfig==2.3.0
|
||||
idna==3.18
|
||||
identify==2.6.19
|
||||
hf-xet==1.5.1
|
||||
h11==0.16.0
|
||||
grpcio==1.81.1
|
||||
gitdb==4.0.12
|
||||
fsspec==2025.10.0
|
||||
frozenlist==1.8.0
|
||||
dill==0.4.0
|
||||
codetiming==1.4.0
|
||||
cloudpickle==3.1.2
|
||||
click==8.4.1
|
||||
charset-normalizer==3.4.7
|
||||
cfgv==3.5.0
|
||||
certifi==2026.5.20
|
||||
attrs==26.1.0
|
||||
annotated-types==0.7.0
|
||||
annotated-doc==0.0.4
|
||||
aiohappyeyeballs==2.6.2
|
||||
absl-py==2.4.0
|
||||
yarl==1.24.2
|
||||
uvicorn==0.40.0
|
||||
tensorboard==2.20.0
|
||||
smart_open==7.6.1
|
||||
sentry-sdk==2.62.0
|
||||
requests==2.34.2
|
||||
referencing==0.37.0
|
||||
pyvers==0.1.0
|
||||
python-discovery==1.4.2
|
||||
python-dateutil==2.9.0.post0
|
||||
pytest==9.1.0
|
||||
pydantic==2.13.4
|
||||
pyasn1_modules==0.4.2
|
||||
proto-plus==1.28.0
|
||||
nvidia-nvjitlink-cu12==12.8.93
|
||||
nvidia-curand-cu12==10.3.9.90
|
||||
omegaconf==2.3.1
|
||||
multiprocess==0.70.18
|
||||
latex2sympy2_extended==1.10.2
|
||||
latex2sympy2==1.5.4
|
||||
googleapis-common-protos==1.75.0
|
||||
cffi==2.0.0
|
||||
anyio==4.13.0
|
||||
aiosignal==1.4.0
|
||||
virtualenv==21.5.0
|
||||
starlette==0.50.0
|
||||
pytest-rerunfailures==16.3
|
||||
pytest-asyncio==1.4.0
|
||||
pandas==2.3.3
|
||||
nvidia-nccl-cu12==2.27.5
|
||||
math-verify==0.8.0
|
||||
jsonschema-specifications==2025.9.1
|
||||
hydra-core==1.3.2
|
||||
huggingface_hub==0.36.2
|
||||
httpx==0.28.1
|
||||
GitPython==3.1.50
|
||||
cryptography==49.0.0
|
||||
aiohttp==3.14.1
|
||||
wandb==0.23.1
|
||||
torchdata==0.11.0
|
||||
numba==0.61.2
|
||||
tensordict==0.10.0
|
||||
pre_commit==4.6.0
|
||||
liger_kernel==0.8.0
|
||||
jsonschema==4.26.0
|
||||
google-auth==2.54.0
|
||||
fastapi==0.127.0
|
||||
aiohttp-cors==0.8.1
|
||||
accelerate==1.12.0
|
||||
llvmlite==0.44.0
|
||||
google-api-core==2.29.0
|
||||
datasets==4.4.2
|
||||
peft==0.18.0
|
||||
opencensus==0.11.4
|
||||
verl==0.7.0.dev0
|
||||
einops==0.8.2
|
||||
py-cpuinfo==9.0.0
|
||||
nvidia-cusparselt-cu12==0.7.1
|
||||
websockets==16.0
|
||||
uvloop==0.22.1
|
||||
sniffio==1.3.1
|
||||
shellingham==1.5.4
|
||||
sentencepiece==0.2.1
|
||||
scipy==1.17.1
|
||||
rignore==0.7.6
|
||||
python-multipart==0.0.32
|
||||
python-json-logger==4.1.0
|
||||
python-dotenv==1.2.2
|
||||
pycountry==26.2.16
|
||||
partial-json-parser==0.2.1.1.post7
|
||||
opentelemetry-semantic-conventions-ai==0.4.13
|
||||
opencv-python-headless==4.13.0.92
|
||||
ninja==1.13.0
|
||||
nest-asyncio==1.6.0
|
||||
msgspec==0.21.1
|
||||
mdurl==0.1.2
|
||||
lark==1.2.2
|
||||
jiter==0.15.0
|
||||
interegular==0.3.3
|
||||
importlib_metadata==8.0.0
|
||||
httptools==0.8.0
|
||||
fastar==0.11.0
|
||||
dnspython==2.8.0
|
||||
distro==1.9.0
|
||||
diskcache==5.6.3
|
||||
detect-installer==0.1.0
|
||||
Deprecated==1.3.1
|
||||
cuda-pathfinder==1.5.5
|
||||
cachetools==7.1.4
|
||||
blake3==1.0.8
|
||||
astor==0.8.1
|
||||
airportsdata==20260315
|
||||
watchfiles==1.2.0
|
||||
tiktoken==0.13.0
|
||||
tilelang==0.1.9
|
||||
markdown-it-py==4.2.0
|
||||
gguf==0.19.0
|
||||
email-validator==2.3.0
|
||||
cuda-tile==1.3.0
|
||||
cupy-cuda12x==14.1.1
|
||||
nvidia-ml-py==13.610.43
|
||||
rich==15.0.0
|
||||
pydantic-settings==2.14.1
|
||||
pydantic-extra-types==2.11.1
|
||||
prometheus-fastapi-instrumentator==7.1.0
|
||||
supervisor==4.3.0
|
||||
tokenspeed-mla==0.1.2
|
||||
outlines_core==0.2.11
|
||||
typer==0.26.7
|
||||
quack-kernels==0.5.0
|
||||
nvidia-nvvm==13.2.78
|
||||
torch_c_dlpack_ext==0.1.5
|
||||
mistral_common==1.11.3
|
||||
fastapi-cloud-cli==0.20.0
|
||||
fastapi-cli==0.0.24
|
||||
cuda-toolkit==13.0.2
|
||||
tabulate==0.10.0
|
||||
sympy==1.14.0
|
||||
setuptools==79.0.1
|
||||
pybase64==1.4.3
|
||||
nvidia-cufile-cu12==1.13.1.3
|
||||
humming-kernels==0.1.4
|
||||
setproctitle==1.3.7
|
||||
triton==3.5.0
|
||||
nvidia-nvtx-cu12==12.8.90
|
||||
nvidia-cusparselt-cu13==0.8.0
|
||||
tokenspeed-triton==3.7.10.post20260531
|
||||
PyJWT==2.13.0
|
||||
pyelftools==0.33
|
||||
nvidia-nvshmem-cu12==3.3.20
|
||||
nvidia-nvtx==13.0.85
|
||||
nvidia-nvshmem-cu13==3.4.5
|
||||
nvidia-nvjitlink==13.0.88
|
||||
nvidia-nccl-cu13==2.28.9
|
||||
nvidia-curand==10.4.0.35
|
||||
nvidia-cufile==1.15.1.6
|
||||
nvidia-cudnn-frontend==1.25.0
|
||||
nvidia-cuda-runtime==13.0.96
|
||||
nvidia-cuda-nvrtc==13.0.88
|
||||
nvidia-cuda-cupti==13.0.85
|
||||
nvidia-cuda-crt==13.3.33
|
||||
nvidia-cuda-cccl==13.3.3.3.1
|
||||
nvidia-cublas==13.1.0.3
|
||||
ml_dtypes==0.5.4
|
||||
loguru==0.7.3
|
||||
jmespath==1.1.0
|
||||
ijson==3.5.0
|
||||
httpx-sse==0.4.3
|
||||
docstring_parser==0.18.0
|
||||
depyf==0.20.0
|
||||
cuda-core==1.0.1
|
||||
cuda-bindings==13.3.1
|
||||
cbor2==6.1.2
|
||||
apache-tvm-ffi==0.1.9
|
||||
mcp==1.27.2
|
||||
opentelemetry-semantic-conventions==0.63b1
|
||||
opentelemetry-proto==1.42.1
|
||||
nvidia-cusparse==12.6.3.3
|
||||
nvidia-cufft==12.0.0.61
|
||||
nvidia-cudnn-cu13==9.19.0.56
|
||||
nvidia-cuda-nvcc==13.2.78
|
||||
cuda-python==13.3.1
|
||||
fastsafetensors==0.3.2
|
||||
tokenizers==0.22.2
|
||||
sse-starlette==3.4.4
|
||||
opentelemetry-sdk==1.42.1
|
||||
nvidia-cutlass-dsl==4.5.2
|
||||
opentelemetry-exporter-otlp-proto-common==1.42.1
|
||||
openai-harmony==0.0.8
|
||||
openai==2.41.1
|
||||
nvidia-cutlass-dsl-libs-base==4.5.2
|
||||
nvidia-cusolver==12.0.4.66
|
||||
nvidia-cuda-tileiras==13.2.78
|
||||
nvidia-cutlass-dsl-libs-cu13==4.5.2
|
||||
lm-format-enforcer==0.11.3
|
||||
ray==2.53.0
|
||||
model-hosting-container-standards==0.1.15
|
||||
opentelemetry-exporter-otlp-proto-http==1.42.1
|
||||
opentelemetry-exporter-otlp-proto-grpc==1.42.1
|
||||
opentelemetry-exporter-otlp==1.42.1
|
||||
opentelemetry-exporter-prometheus==0.63b1
|
||||
xgrammar==0.1.27
|
||||
torchvision==0.24.0
|
||||
torchaudio==2.9.0
|
||||
flashinfer-python==0.5.3
|
||||
compressed-tensors==0.12.2
|
||||
vllm==0.12.0
|
||||
flash_attn==2.8.3
|
||||
flashinfer-cubin==0.5.3
|
||||
pyparsing==3.3.2
|
||||
kiwisolver==1.5.0
|
||||
fonttools==4.63.0
|
||||
cycler==0.12.1
|
||||
contourpy==1.3.3
|
||||
matplotlib==3.11.0
|
||||
109
artifacts/wandb-metadata.json
Normal file
109
artifacts/wandb-metadata.json
Normal file
@@ -0,0 +1,109 @@
|
||||
{
|
||||
"os": "Linux-6.8.0-71-generic-x86_64-with-glibc2.36",
|
||||
"python": "CPython 3.12.13",
|
||||
"startedAt": "2026-07-03T03:20:40.327312Z",
|
||||
"args": [
|
||||
"--node-ip-address=198.19.44.212",
|
||||
"--node-manager-port=36871",
|
||||
"--object-store-name=/tmp/ray_q3g_physics_grpo_Qwen3_8B/session_2026-07-03_03-18-52_070071_2279047/sockets/plasma_store",
|
||||
"--raylet-name=/tmp/ray_q3g_physics_grpo_Qwen3_8B/session_2026-07-03_03-18-52_070071_2279047/sockets/raylet",
|
||||
"--redis-address=None",
|
||||
"--metrics-agent-port=56405",
|
||||
"--logging-rotate-bytes=536870912",
|
||||
"--logging-rotate-backup-count=5",
|
||||
"--runtime-env-agent-port=64277",
|
||||
"--gcs-address=198.19.44.212:64574",
|
||||
"--session-name=session_2026-07-03_03-18-52_070071_2279047",
|
||||
"--temp-dir=/tmp/ray_q3g_physics_grpo_Qwen3_8B",
|
||||
"--webui=",
|
||||
"--cluster-id=d7576956015edcd024cf2ca706acce8be3fb6519bb7b0c87bfccae89",
|
||||
"--startup-token=32",
|
||||
"--worker-launch-time-ms=1783048735496",
|
||||
"--node-id=e88e44745be3f8c4449808648428760387c2aa53671532e80c9f13b3",
|
||||
"--runtime-env-hash=1652680156"
|
||||
],
|
||||
"program": "/mnt/mole/SDPO/L2T/.venv/lib/python3.12/site-packages/ray/_private/workers/default_worker.py",
|
||||
"codePath": ".venv/lib/python3.12/site-packages/ray/_private/workers/default_worker.py",
|
||||
"codePathLocal": ".venv/lib/python3.12/site-packages/ray/_private/workers/default_worker.py",
|
||||
"git": {
|
||||
"remote": "https://github.com/jungseongryong/L2T.git",
|
||||
"commit": "ee6a667700697069b93db636aa937f970c1fb57a"
|
||||
},
|
||||
"email": "jungsr1116@cau.ac.kr",
|
||||
"root": "/mnt/mole/SDPO/L2T",
|
||||
"host": "mole-workspace-rw",
|
||||
"executable": "/mnt/mole/SDPO/L2T/.venv/bin/python",
|
||||
"cpu_count": 64,
|
||||
"cpu_count_logical": 128,
|
||||
"gpu": "NVIDIA H200",
|
||||
"gpu_count": 8,
|
||||
"disk": {
|
||||
"/": {
|
||||
"total": "46086056050688",
|
||||
"used": "25787742543872"
|
||||
}
|
||||
},
|
||||
"memory": {
|
||||
"total": "2163980390400"
|
||||
},
|
||||
"gpu_nvidia": [
|
||||
{
|
||||
"name": "NVIDIA H200",
|
||||
"memoryTotal": "150754820096",
|
||||
"cudaCores": 16896,
|
||||
"architecture": "Hopper",
|
||||
"uuid": "GPU-c0451621-891c-7169-976d-71d81958db34"
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA H200",
|
||||
"memoryTotal": "150754820096",
|
||||
"cudaCores": 16896,
|
||||
"architecture": "Hopper",
|
||||
"uuid": "GPU-e2a29dfa-7dad-6cd7-fb05-79d0060f27f0"
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA H200",
|
||||
"memoryTotal": "150754820096",
|
||||
"cudaCores": 16896,
|
||||
"architecture": "Hopper",
|
||||
"uuid": "GPU-d69bb741-f2a7-a375-7f56-5c0d09d797b2"
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA H200",
|
||||
"memoryTotal": "150754820096",
|
||||
"cudaCores": 16896,
|
||||
"architecture": "Hopper",
|
||||
"uuid": "GPU-1bbbea1e-7541-3126-1bfc-51b43a67577e"
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA H200",
|
||||
"memoryTotal": "150754820096",
|
||||
"cudaCores": 16896,
|
||||
"architecture": "Hopper",
|
||||
"uuid": "GPU-6e0055af-452d-3f46-ce21-0c71e347a608"
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA H200",
|
||||
"memoryTotal": "150754820096",
|
||||
"cudaCores": 16896,
|
||||
"architecture": "Hopper",
|
||||
"uuid": "GPU-317fdb5b-b280-71d8-398e-a843ad897437"
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA H200",
|
||||
"memoryTotal": "150754820096",
|
||||
"cudaCores": 16896,
|
||||
"architecture": "Hopper",
|
||||
"uuid": "GPU-f19d8161-d2dd-49a1-dcb8-5c341b5e4514"
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA H200",
|
||||
"memoryTotal": "150754820096",
|
||||
"cudaCores": 16896,
|
||||
"architecture": "Hopper",
|
||||
"uuid": "GPU-961f5ebf-4bf8-6855-828b-c065af57b87c"
|
||||
}
|
||||
],
|
||||
"cudaVersion": "13.0",
|
||||
"writerId": "1n1ucg63toheyodgdd2dcwkutcbcxhn8"
|
||||
}
|
||||
1
artifacts/wandb-summary.json
Normal file
1
artifacts/wandb-summary.json
Normal file
File diff suppressed because one or more lines are too long
Reference in New Issue
Block a user