初始化项目,由ModelHub XC社区提供模型
Model: promotion/ronpo-qwen3-8b-fair-inpo-avg-s42 Source: Original Platform
This commit is contained in:
36
.gitattributes
vendored
Normal file
36
.gitattributes
vendored
Normal file
@@ -0,0 +1,36 @@
|
|||||||
|
*.7z filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.arrow filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.bin filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.ftz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.gz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.h5 filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.joblib filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.model filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.npy filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.npz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.onnx filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.ot filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.parquet filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pb filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pickle filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pkl filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pt filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pth filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.rar filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||||
|
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tar filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tflite filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tgz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.wasm filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.xz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.zip filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.zst filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
||||||
|
tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
||||||
12
README.md
Normal file
12
README.md
Normal file
@@ -0,0 +1,12 @@
|
|||||||
|
---
|
||||||
|
base_model: Qwen/Qwen3-8B
|
||||||
|
library_name: transformers
|
||||||
|
tags:
|
||||||
|
- preference-optimization
|
||||||
|
- ronpo
|
||||||
|
- qwen3
|
||||||
|
---
|
||||||
|
|
||||||
|
# Qwen3-8B fair-demo checkpoint: inpo_avg
|
||||||
|
|
||||||
|
Validation-selected candidate: `inpo_b`. Exact evaluator, selection, sweep, and training metadata are stored under `experiment/`.
|
||||||
9
all_results.json
Normal file
9
all_results.json
Normal file
@@ -0,0 +1,9 @@
|
|||||||
|
{
|
||||||
|
"epoch": 0.8599068434252956,
|
||||||
|
"total_flos": 0.0,
|
||||||
|
"train_loss": 24.184869588216145,
|
||||||
|
"train_runtime": 7109.0777,
|
||||||
|
"train_samples": 16746,
|
||||||
|
"train_samples_per_second": 2.026,
|
||||||
|
"train_steps_per_second": 0.127
|
||||||
|
}
|
||||||
89
chat_template.jinja
Normal file
89
chat_template.jinja
Normal file
@@ -0,0 +1,89 @@
|
|||||||
|
{%- if tools %}
|
||||||
|
{{- '<|im_start|>system\n' }}
|
||||||
|
{%- if messages[0].role == 'system' %}
|
||||||
|
{{- messages[0].content + '\n\n' }}
|
||||||
|
{%- endif %}
|
||||||
|
{{- "# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
|
||||||
|
{%- for tool in tools %}
|
||||||
|
{{- "\n" }}
|
||||||
|
{{- tool | tojson }}
|
||||||
|
{%- endfor %}
|
||||||
|
{{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
|
||||||
|
{%- else %}
|
||||||
|
{%- if messages[0].role == 'system' %}
|
||||||
|
{{- '<|im_start|>system\n' + messages[0].content + '<|im_end|>\n' }}
|
||||||
|
{%- endif %}
|
||||||
|
{%- endif %}
|
||||||
|
{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
|
||||||
|
{%- for message in messages[::-1] %}
|
||||||
|
{%- set index = (messages|length - 1) - loop.index0 %}
|
||||||
|
{%- if ns.multi_step_tool and message.role == "user" and message.content is string and not(message.content.startswith('<tool_response>') and message.content.endswith('</tool_response>')) %}
|
||||||
|
{%- set ns.multi_step_tool = false %}
|
||||||
|
{%- set ns.last_query_index = index %}
|
||||||
|
{%- endif %}
|
||||||
|
{%- endfor %}
|
||||||
|
{%- for message in messages %}
|
||||||
|
{%- if message.content is string %}
|
||||||
|
{%- set content = message.content %}
|
||||||
|
{%- else %}
|
||||||
|
{%- set content = '' %}
|
||||||
|
{%- endif %}
|
||||||
|
{%- if (message.role == "user") or (message.role == "system" and not loop.first) %}
|
||||||
|
{{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
|
||||||
|
{%- elif message.role == "assistant" %}
|
||||||
|
{%- set reasoning_content = '' %}
|
||||||
|
{%- if message.reasoning_content is string %}
|
||||||
|
{%- set reasoning_content = message.reasoning_content %}
|
||||||
|
{%- else %}
|
||||||
|
{%- if '</think>' in content %}
|
||||||
|
{%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
|
||||||
|
{%- set content = content.split('</think>')[-1].lstrip('\n') %}
|
||||||
|
{%- endif %}
|
||||||
|
{%- endif %}
|
||||||
|
{%- if loop.index0 > ns.last_query_index %}
|
||||||
|
{%- if loop.last or (not loop.last and reasoning_content) %}
|
||||||
|
{{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content.strip('\n') + '\n</think>\n\n' + content.lstrip('\n') }}
|
||||||
|
{%- else %}
|
||||||
|
{{- '<|im_start|>' + message.role + '\n' + content }}
|
||||||
|
{%- endif %}
|
||||||
|
{%- else %}
|
||||||
|
{{- '<|im_start|>' + message.role + '\n' + content }}
|
||||||
|
{%- endif %}
|
||||||
|
{%- if message.tool_calls %}
|
||||||
|
{%- for tool_call in message.tool_calls %}
|
||||||
|
{%- if (loop.first and content) or (not loop.first) %}
|
||||||
|
{{- '\n' }}
|
||||||
|
{%- endif %}
|
||||||
|
{%- if tool_call.function %}
|
||||||
|
{%- set tool_call = tool_call.function %}
|
||||||
|
{%- endif %}
|
||||||
|
{{- '<tool_call>\n{"name": "' }}
|
||||||
|
{{- tool_call.name }}
|
||||||
|
{{- '", "arguments": ' }}
|
||||||
|
{%- if tool_call.arguments is string %}
|
||||||
|
{{- tool_call.arguments }}
|
||||||
|
{%- else %}
|
||||||
|
{{- tool_call.arguments | tojson }}
|
||||||
|
{%- endif %}
|
||||||
|
{{- '}\n</tool_call>' }}
|
||||||
|
{%- endfor %}
|
||||||
|
{%- endif %}
|
||||||
|
{{- '<|im_end|>\n' }}
|
||||||
|
{%- elif message.role == "tool" %}
|
||||||
|
{%- if loop.first or (messages[loop.index0 - 1].role != "tool") %}
|
||||||
|
{{- '<|im_start|>user' }}
|
||||||
|
{%- endif %}
|
||||||
|
{{- '\n<tool_response>\n' }}
|
||||||
|
{{- content }}
|
||||||
|
{{- '\n</tool_response>' }}
|
||||||
|
{%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
|
||||||
|
{{- '<|im_end|>\n' }}
|
||||||
|
{%- endif %}
|
||||||
|
{%- endif %}
|
||||||
|
{%- endfor %}
|
||||||
|
{%- if add_generation_prompt %}
|
||||||
|
{{- '<|im_start|>assistant\n' }}
|
||||||
|
{%- if enable_thinking is defined and enable_thinking is false %}
|
||||||
|
{{- '<think>\n\n</think>\n\n' }}
|
||||||
|
{%- endif %}
|
||||||
|
{%- endif %}
|
||||||
71
config.json
Normal file
71
config.json
Normal file
@@ -0,0 +1,71 @@
|
|||||||
|
{
|
||||||
|
"architectures": [
|
||||||
|
"Qwen3ForCausalLM"
|
||||||
|
],
|
||||||
|
"attention_bias": false,
|
||||||
|
"attention_dropout": 0.0,
|
||||||
|
"bos_token_id": null,
|
||||||
|
"dtype": "bfloat16",
|
||||||
|
"eos_token_id": 151645,
|
||||||
|
"head_dim": 128,
|
||||||
|
"hidden_act": "silu",
|
||||||
|
"hidden_size": 4096,
|
||||||
|
"initializer_range": 0.02,
|
||||||
|
"intermediate_size": 12288,
|
||||||
|
"layer_types": [
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention"
|
||||||
|
],
|
||||||
|
"max_position_embeddings": 40960,
|
||||||
|
"max_window_layers": 36,
|
||||||
|
"model_type": "qwen3",
|
||||||
|
"num_attention_heads": 32,
|
||||||
|
"num_hidden_layers": 36,
|
||||||
|
"num_key_value_heads": 8,
|
||||||
|
"pad_token_id": 151643,
|
||||||
|
"rms_norm_eps": 1e-06,
|
||||||
|
"rope_parameters": {
|
||||||
|
"rope_theta": 1000000,
|
||||||
|
"rope_type": "default"
|
||||||
|
},
|
||||||
|
"sliding_window": null,
|
||||||
|
"tie_word_embeddings": false,
|
||||||
|
"transformers_version": "5.13.0",
|
||||||
|
"use_cache": true,
|
||||||
|
"use_sliding_window": false,
|
||||||
|
"vocab_size": 151936
|
||||||
|
}
|
||||||
58
config.yaml
Normal file
58
config.yaml
Normal file
@@ -0,0 +1,58 @@
|
|||||||
|
model_name_or_path: /NHNHOME/26msit001_A/BASE/AIPR/sjkim/revision_qwen3_8b/full_iter1/flagship_20260712/cache/huggingface/hub/models--Qwen--Qwen3-8B/snapshots/b968826d9c46dd6066d109eabc6255188de91218
|
||||||
|
torch_dtype: null
|
||||||
|
attn_implementation: sdpa
|
||||||
|
dataset_mixer:
|
||||||
|
/NHNHOME/26msit001_A/BASE/AIPR/sjkim/revision_qwen3_8b/full_iter1/flagship_20260712/precomputed/avg: 1.0
|
||||||
|
dataset_splits:
|
||||||
|
- train
|
||||||
|
- test
|
||||||
|
preprocessing_num_workers: 4
|
||||||
|
bf16: true
|
||||||
|
loss_type: inpo
|
||||||
|
eta: 0.1
|
||||||
|
ratio: 0.3333
|
||||||
|
max_history_t: 1
|
||||||
|
history_weights:
|
||||||
|
- 1.0
|
||||||
|
dpo_beta: 0.05
|
||||||
|
simpo_beta: 2.0
|
||||||
|
simpo_gamma: 0.6
|
||||||
|
ronpo_alpha: 0.5
|
||||||
|
ronpo_tau: 0.05
|
||||||
|
ronpo_target_column: ronpo_target
|
||||||
|
reference_anchor_weight: 0.075
|
||||||
|
preference_sft_weight: 0.0075
|
||||||
|
ht_target_column: ht_target
|
||||||
|
ht_target_scale: 1.0
|
||||||
|
beta: 10.0
|
||||||
|
learning_rate: 5.0e-07
|
||||||
|
lr_scheduler_type: cosine
|
||||||
|
warmup_ratio: 0.2
|
||||||
|
optim: adamw_torch
|
||||||
|
weight_decay: 0.0
|
||||||
|
max_grad_norm: 1.0
|
||||||
|
seed: 42
|
||||||
|
gradient_accumulation_steps: 16
|
||||||
|
gradient_checkpointing: true
|
||||||
|
num_train_epochs: 1
|
||||||
|
max_steps: 900
|
||||||
|
per_device_train_batch_size: 1
|
||||||
|
per_device_eval_batch_size: 1
|
||||||
|
max_length: 2048
|
||||||
|
max_prompt_length: 1800
|
||||||
|
do_eval: false
|
||||||
|
eval_strategy: 'no'
|
||||||
|
logging_steps: 5
|
||||||
|
log_level: info
|
||||||
|
generate_during_eval: false
|
||||||
|
load_best_model_at_end: false
|
||||||
|
save_strategy: steps
|
||||||
|
save_steps: 900
|
||||||
|
save_total_limit: 1
|
||||||
|
save_only_model: true
|
||||||
|
save_safetensors: true
|
||||||
|
push_to_hub: false
|
||||||
|
report_to:
|
||||||
|
- wandb
|
||||||
|
output_dir: /NHNHOME/26msit001_A/BASE/AIPR/sjkim/revision_qwen3_8b/full_iter1/fair_demo_20260715/sweep/candidates/inpo_b
|
||||||
|
run_name: qwen3-8b-fair-demo-inpo_b
|
||||||
85
experiment/evaluator_lock.json
Normal file
85
experiment/evaluator_lock.json
Normal file
@@ -0,0 +1,85 @@
|
|||||||
|
{
|
||||||
|
"status": "LOCKED_BEFORE_ANY_NEW_METHOD_RANKING",
|
||||||
|
"locked_at": "2026-07-15T20:20:29+09:00",
|
||||||
|
"objective_signals": [
|
||||||
|
"skywork",
|
||||||
|
"armo_safety",
|
||||||
|
"length_conciseness"
|
||||||
|
],
|
||||||
|
"objective_semantics": [
|
||||||
|
"helpfulness",
|
||||||
|
"safety",
|
||||||
|
"conciseness"
|
||||||
|
],
|
||||||
|
"primary": {
|
||||||
|
"name": "open_weight_panel_mean_prompt_worst_vs_base",
|
||||||
|
"definition": "For each prompt and objective, average win=1/tie=0.5/loss=0 over two A/B positions and two judges; take the minimum objective, then mean over prompts.",
|
||||||
|
"judges": [
|
||||||
|
{
|
||||||
|
"model": "openai/gpt-oss-120b",
|
||||||
|
"revision": "b5c939de8f754692c1647ca79fbf85e8c1e70f8a"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"model": "Qwen/Qwen3-32B",
|
||||||
|
"revision": "9216db5781bf21249d130ec9da846c4624c16137"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"decode": {
|
||||||
|
"temperature": 0.0,
|
||||||
|
"top_p": 1.0,
|
||||||
|
"max_new_tokens": 512,
|
||||||
|
"seed": 42
|
||||||
|
},
|
||||||
|
"position_swap": true,
|
||||||
|
"validation_selection_rule": "Highest eligible mean prompt-level worst panel score within each method; an exact numeric tie is broken by candidate_id lexical order."
|
||||||
|
},
|
||||||
|
"secondary": {
|
||||||
|
"normalization": "For each locked signal, divide each evaluation prompt's candidate-minus-base raw delta by the population SD of the diagnostic base and matched-control scores; take the prompt-level minimum across objectives, then the mean. No per-prompt min-max.",
|
||||||
|
"diagnostic_control_scale": {
|
||||||
|
"skywork": 22.807301785782975,
|
||||||
|
"armo_safety": 0.3098083353203591,
|
||||||
|
"length_conciseness": 1.5891969646897472
|
||||||
|
},
|
||||||
|
"reward_signal_provenance": {
|
||||||
|
"skywork": {
|
||||||
|
"model": "Skywork/Skywork-Reward-V2-Llama-3.1-8B",
|
||||||
|
"revision": "cba2f842f3f1af2f1b2f0d35e794d789976390c5",
|
||||||
|
"semantics": "helpfulness"
|
||||||
|
},
|
||||||
|
"armo_safety": {
|
||||||
|
"model": "RLHFlow/ArmoRM-Llama3-8B-v0.1",
|
||||||
|
"revision": "eb2676d20da2f2d41082289d23c59b9f7427f955",
|
||||||
|
"head": "beavertails-is_safe",
|
||||||
|
"transform": "identity"
|
||||||
|
},
|
||||||
|
"length_conciseness": {
|
||||||
|
"deterministic": "-log1p(response_word_count)",
|
||||||
|
"tokenization": "Python str.split whitespace words"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"report_raw_paired_deltas": true
|
||||||
|
},
|
||||||
|
"bootstrap": {
|
||||||
|
"paired_prompt_resamples": 2000,
|
||||||
|
"seed": 42,
|
||||||
|
"interval": "percentile_95"
|
||||||
|
},
|
||||||
|
"power": {
|
||||||
|
"target_absolute_effect": 0.05,
|
||||||
|
"paired_sd": 0.13831629563357775,
|
||||||
|
"required_prompts_80pct_power": 61,
|
||||||
|
"planned_fresh_test_prompts": 1024,
|
||||||
|
"alpha_two_sided": 0.05,
|
||||||
|
"power": 0.8
|
||||||
|
},
|
||||||
|
"fresh_test_source": {
|
||||||
|
"dataset": "HuggingFaceH4/ultrachat_200k",
|
||||||
|
"revision": "8049631c405ae6576f93f445c6b8166f76f5505a",
|
||||||
|
"split": "test_sft"
|
||||||
|
},
|
||||||
|
"diagnostic_summary_sha256": "e365ed814926ac5c17777e0fc65d0299a45b81bd390542889bbd55b5453d4ffe",
|
||||||
|
"judge_diagnostic_lock_sha256": "fce2d97806053cabdf3d358806f4baa80df7f89d71d8c1c298883327d4e02635",
|
||||||
|
"prereg_sha256": "71cbd1ac17c3e807f2ece1cb0a82374bbe74edb654c2845b97c3222ed680f0d0",
|
||||||
|
"method_ranking_computed": false,
|
||||||
|
"spent_sealed_split_touched": false
|
||||||
|
}
|
||||||
99
experiment/selection_lock.json
Normal file
99
experiment/selection_lock.json
Normal file
@@ -0,0 +1,99 @@
|
|||||||
|
{
|
||||||
|
"status": "VALIDATION_SELECTION_LOCKED_BEFORE_FRESH_TEST",
|
||||||
|
"locked_at": "2026-07-15T20:37:11+09:00",
|
||||||
|
"selection_metric": "open_weight_panel_mean_prompt_worst_vs_base",
|
||||||
|
"tie_break": "candidate_id lexical order",
|
||||||
|
"selected_by_method": {
|
||||||
|
"dpo": {
|
||||||
|
"candidate_id": "dpo_b",
|
||||||
|
"validation_primary": 0.40625,
|
||||||
|
"validation_primary_ci95": [
|
||||||
|
0.3837890625,
|
||||||
|
0.427734375
|
||||||
|
],
|
||||||
|
"eligible_candidates": [
|
||||||
|
"dpo_a",
|
||||||
|
"dpo_b"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"ht_mnpo_helpfulness": {
|
||||||
|
"candidate_id": "ht_help_b",
|
||||||
|
"validation_primary": 0.4072265625,
|
||||||
|
"validation_primary_ci95": [
|
||||||
|
0.3828125,
|
||||||
|
0.4296875
|
||||||
|
],
|
||||||
|
"eligible_candidates": [
|
||||||
|
"ht_help_b"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"ht_mnpo_safety": {
|
||||||
|
"candidate_id": "ht_safety_a",
|
||||||
|
"validation_primary": 0.4150390625,
|
||||||
|
"validation_primary_ci95": [
|
||||||
|
0.3925537109375,
|
||||||
|
0.4384765625
|
||||||
|
],
|
||||||
|
"eligible_candidates": [
|
||||||
|
"ht_safety_a",
|
||||||
|
"ht_safety_b"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"inpo_avg": {
|
||||||
|
"candidate_id": "inpo_b",
|
||||||
|
"validation_primary": 0.40625,
|
||||||
|
"validation_primary_ci95": [
|
||||||
|
0.3818359375,
|
||||||
|
0.427734375
|
||||||
|
],
|
||||||
|
"eligible_candidates": [
|
||||||
|
"inpo_b"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"ronpo_full_expect": {
|
||||||
|
"candidate_id": "ronpo_full_a",
|
||||||
|
"validation_primary": 0.4033203125,
|
||||||
|
"validation_primary_ci95": [
|
||||||
|
0.37890625,
|
||||||
|
0.42580566406249987
|
||||||
|
],
|
||||||
|
"eligible_candidates": [
|
||||||
|
"ronpo_full_a",
|
||||||
|
"ronpo_full_b"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"ronpo_k_only": {
|
||||||
|
"candidate_id": "ronpo_top_a",
|
||||||
|
"validation_primary": 0.412109375,
|
||||||
|
"validation_primary_ci95": [
|
||||||
|
0.3876953125,
|
||||||
|
0.43557128906249987
|
||||||
|
],
|
||||||
|
"eligible_candidates": [
|
||||||
|
"ronpo_top_a"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"sppo_avg": {
|
||||||
|
"candidate_id": "sppo_b",
|
||||||
|
"validation_primary": 0.4111328125,
|
||||||
|
"validation_primary_ci95": [
|
||||||
|
0.3866943359375,
|
||||||
|
0.4345703125
|
||||||
|
],
|
||||||
|
"eligible_candidates": [
|
||||||
|
"sppo_b"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"selected_ronpo_overall": "ronpo_top_a",
|
||||||
|
"failed_methods": [
|
||||||
|
"ht_mnpo_conciseness",
|
||||||
|
"ipo",
|
||||||
|
"simpo"
|
||||||
|
],
|
||||||
|
"panel_summary_sha256": "8850f23e16cadf56a99e3d562d224ce5337ab36fa588d8a843382853e6ad17bd",
|
||||||
|
"evaluator_lock_sha256": "ab7b12c84932ed57acfec7edbcbc1ff1df5e9acd203869b1458343adf6e5b2c8",
|
||||||
|
"grid_sha256": "abfd982ef10c3bc71364752241150286ebf06be5e97f087596adf80e55d459e0",
|
||||||
|
"fresh_test_opened": false,
|
||||||
|
"spent_sealed_split_touched": false
|
||||||
|
}
|
||||||
46
experiment/sweep_grid.json
Normal file
46
experiment/sweep_grid.json
Normal file
@@ -0,0 +1,46 @@
|
|||||||
|
{
|
||||||
|
"schema_version": 1,
|
||||||
|
"status": "frozen_before_training_and_validation_ranking",
|
||||||
|
"frozen_at_kst": "2026-07-15T09:01:45+09:00",
|
||||||
|
"seed": 42,
|
||||||
|
"common": {
|
||||||
|
"optimizer_steps": 900,
|
||||||
|
"effective_batch_size": 16,
|
||||||
|
"per_device_train_batch_size": 1,
|
||||||
|
"gradient_accumulation_steps": 16,
|
||||||
|
"gradient_checkpointing": true,
|
||||||
|
"bf16": true,
|
||||||
|
"attn_implementation": "sdpa",
|
||||||
|
"cudnn_sdpa": false,
|
||||||
|
"lr_scheduler_type": "cosine",
|
||||||
|
"max_length": 2048,
|
||||||
|
"max_prompt_length": 1800,
|
||||||
|
"save_total_limit": 1,
|
||||||
|
"wandb_entity": "promotion-kim",
|
||||||
|
"wandb_project": "mnpo",
|
||||||
|
"wandb_group": "qwen3-8b-fair-demo-20260715"
|
||||||
|
},
|
||||||
|
"budget_rule": "Exactly two configs per reported method; no extension after validation rankings are visible.",
|
||||||
|
"candidates": [
|
||||||
|
{"id":"ronpo_full_a","method":"ronpo_full_expect","dataset":"ronpo_full_expect_kall","learning_rate":1e-7,"warmup_ratio":0.15,"ronpo_alpha":0.25,"ronpo_tau":0.10,"reference_anchor_weight":0.075,"preference_sft_weight":0.0075,"expected_support_k":0},
|
||||||
|
{"id":"ronpo_full_b","method":"ronpo_full_expect","dataset":"ronpo_full_expect_k6","learning_rate":2.5e-7,"warmup_ratio":0.20,"ronpo_alpha":0.35,"ronpo_tau":0.05,"reference_anchor_weight":0.10,"preference_sft_weight":0.01,"expected_support_k":6},
|
||||||
|
{"id":"ronpo_top_a","method":"ronpo_k_only","dataset":"ronpo_k_only_k1","learning_rate":5e-8,"warmup_ratio":0.20,"ronpo_alpha":0.15,"ronpo_tau":0.10,"reference_anchor_weight":0.10,"preference_sft_weight":0.01,"expected_support_k":1},
|
||||||
|
{"id":"ronpo_top_b","method":"ronpo_k_only","dataset":"ronpo_k_only_k2","learning_rate":1e-7,"warmup_ratio":0.15,"ronpo_alpha":0.25,"ronpo_tau":0.10,"reference_anchor_weight":0.075,"preference_sft_weight":0.0075,"expected_support_k":2},
|
||||||
|
{"id":"dpo_a","method":"dpo","dataset":"avg","learning_rate":2.5e-7,"warmup_ratio":0.10,"dpo_beta":0.05,"reference_anchor_weight":0.02,"preference_sft_weight":0.002},
|
||||||
|
{"id":"dpo_b","method":"dpo","dataset":"avg","learning_rate":5e-7,"warmup_ratio":0.15,"dpo_beta":0.10,"reference_anchor_weight":0.05,"preference_sft_weight":0.005},
|
||||||
|
{"id":"ipo_a","method":"ipo","dataset":"avg","learning_rate":2.5e-7,"warmup_ratio":0.15,"dpo_beta":0.20,"reference_anchor_weight":0.05,"preference_sft_weight":0.005},
|
||||||
|
{"id":"ipo_b","method":"ipo","dataset":"avg","learning_rate":5e-7,"warmup_ratio":0.20,"dpo_beta":0.50,"reference_anchor_weight":0.075,"preference_sft_weight":0.0075},
|
||||||
|
{"id":"simpo_a","method":"simpo","dataset":"avg","learning_rate":2.5e-7,"warmup_ratio":0.10,"simpo_beta":1.0,"simpo_gamma":0.3,"reference_anchor_weight":0.02,"preference_sft_weight":0.002},
|
||||||
|
{"id":"simpo_b","method":"simpo","dataset":"avg","learning_rate":5e-7,"warmup_ratio":0.15,"simpo_beta":2.0,"simpo_gamma":0.6,"reference_anchor_weight":0.05,"preference_sft_weight":0.005},
|
||||||
|
{"id":"sppo_a","method":"sppo_avg","dataset":"avg","learning_rate":2.5e-7,"warmup_ratio":0.15,"eta":0.005,"reference_anchor_weight":0.05,"preference_sft_weight":0.005},
|
||||||
|
{"id":"sppo_b","method":"sppo_avg","dataset":"avg","learning_rate":5e-7,"warmup_ratio":0.20,"eta":0.010,"reference_anchor_weight":0.075,"preference_sft_weight":0.0075},
|
||||||
|
{"id":"inpo_a","method":"inpo_avg","dataset":"avg","learning_rate":2.5e-7,"warmup_ratio":0.15,"eta":0.050,"ratio":0.3333,"reference_anchor_weight":0.05,"preference_sft_weight":0.005},
|
||||||
|
{"id":"inpo_b","method":"inpo_avg","dataset":"avg","learning_rate":5e-7,"warmup_ratio":0.20,"eta":0.100,"ratio":0.3333,"reference_anchor_weight":0.075,"preference_sft_weight":0.0075},
|
||||||
|
{"id":"ht_help_a","method":"ht_mnpo_helpfulness","dataset":"ht_mnpo_helpfulness","learning_rate":2.5e-7,"warmup_ratio":0.10,"eta":0.005,"reference_anchor_weight":0.02,"preference_sft_weight":0.002},
|
||||||
|
{"id":"ht_help_b","method":"ht_mnpo_helpfulness","dataset":"ht_mnpo_helpfulness","learning_rate":5e-7,"warmup_ratio":0.15,"eta":0.010,"reference_anchor_weight":0.05,"preference_sft_weight":0.005},
|
||||||
|
{"id":"ht_safety_a","method":"ht_mnpo_safety","dataset":"ht_mnpo_safety","learning_rate":2.5e-7,"warmup_ratio":0.10,"eta":0.005,"reference_anchor_weight":0.02,"preference_sft_weight":0.002},
|
||||||
|
{"id":"ht_safety_b","method":"ht_mnpo_safety","dataset":"ht_mnpo_safety","learning_rate":5e-7,"warmup_ratio":0.15,"eta":0.010,"reference_anchor_weight":0.05,"preference_sft_weight":0.005},
|
||||||
|
{"id":"ht_concise_a","method":"ht_mnpo_conciseness","dataset":"ht_mnpo_conciseness","learning_rate":2.5e-7,"warmup_ratio":0.10,"eta":0.005,"reference_anchor_weight":0.02,"preference_sft_weight":0.002},
|
||||||
|
{"id":"ht_concise_b","method":"ht_mnpo_conciseness","dataset":"ht_mnpo_conciseness","learning_rate":5e-7,"warmup_ratio":0.15,"eta":0.010,"reference_anchor_weight":0.05,"preference_sft_weight":0.005}
|
||||||
|
]
|
||||||
|
}
|
||||||
20
experiment/training_status.json
Normal file
20
experiment/training_status.json
Normal file
@@ -0,0 +1,20 @@
|
|||||||
|
{
|
||||||
|
"status": "completed",
|
||||||
|
"candidate_id": "inpo_b",
|
||||||
|
"method": "inpo_avg",
|
||||||
|
"seed": 42,
|
||||||
|
"gpu": 0,
|
||||||
|
"pid": 1079779,
|
||||||
|
"config": "/NHNHOME/26msit001_A/BASE/AIPR/sjkim/revision_qwen3_8b/full_iter1/fair_demo_20260715/sweep/candidates/inpo_b/config.yaml",
|
||||||
|
"dataset": "/NHNHOME/26msit001_A/BASE/AIPR/sjkim/revision_qwen3_8b/full_iter1/flagship_20260712/precomputed/avg",
|
||||||
|
"optimizer_steps": 900,
|
||||||
|
"effective_batch_size": 16,
|
||||||
|
"wandb_run_id": "098086245ba4",
|
||||||
|
"wandb_url": "https://wandb.ai/promotion-kim/mnpo/runs/098086245ba4",
|
||||||
|
"started_at": "2026-07-15T15:36:21+09:00",
|
||||||
|
"returncode": 0,
|
||||||
|
"model_complete": true,
|
||||||
|
"finite_results": true,
|
||||||
|
"measured_step": null,
|
||||||
|
"completed_at": "2026-07-15T17:35:44+09:00"
|
||||||
|
}
|
||||||
12
generation_config.json
Normal file
12
generation_config.json
Normal file
@@ -0,0 +1,12 @@
|
|||||||
|
{
|
||||||
|
"do_sample": true,
|
||||||
|
"eos_token_id": [
|
||||||
|
151645,
|
||||||
|
151643
|
||||||
|
],
|
||||||
|
"pad_token_id": 151643,
|
||||||
|
"temperature": 0.6,
|
||||||
|
"top_k": 20,
|
||||||
|
"top_p": 0.95,
|
||||||
|
"transformers_version": "5.13.0"
|
||||||
|
}
|
||||||
3
model.safetensors
Normal file
3
model.safetensors
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:13ecf1fe52b368464f8bba1caa7719d0c7a3e6b555c26ef3872918666fcd836c
|
||||||
|
size 16381517208
|
||||||
3
tokenizer.json
Normal file
3
tokenizer.json
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:be75606093db2094d7cd20f3c2f385c212750648bd6ea4fb2bf507a6a4c55506
|
||||||
|
size 11422650
|
||||||
30
tokenizer_config.json
Normal file
30
tokenizer_config.json
Normal file
@@ -0,0 +1,30 @@
|
|||||||
|
{
|
||||||
|
"add_prefix_space": false,
|
||||||
|
"backend": "tokenizers",
|
||||||
|
"bos_token": null,
|
||||||
|
"clean_up_tokenization_spaces": false,
|
||||||
|
"eos_token": "<|im_end|>",
|
||||||
|
"errors": "replace",
|
||||||
|
"extra_special_tokens": [
|
||||||
|
"<|im_start|>",
|
||||||
|
"<|im_end|>",
|
||||||
|
"<|object_ref_start|>",
|
||||||
|
"<|object_ref_end|>",
|
||||||
|
"<|box_start|>",
|
||||||
|
"<|box_end|>",
|
||||||
|
"<|quad_start|>",
|
||||||
|
"<|quad_end|>",
|
||||||
|
"<|vision_start|>",
|
||||||
|
"<|vision_end|>",
|
||||||
|
"<|vision_pad|>",
|
||||||
|
"<|image_pad|>",
|
||||||
|
"<|video_pad|>"
|
||||||
|
],
|
||||||
|
"is_local": true,
|
||||||
|
"local_files_only": true,
|
||||||
|
"model_max_length": 2048,
|
||||||
|
"pad_token": "<|endoftext|>",
|
||||||
|
"split_special_tokens": false,
|
||||||
|
"tokenizer_class": "Qwen2Tokenizer",
|
||||||
|
"unk_token": null
|
||||||
|
}
|
||||||
9
train_results.json
Normal file
9
train_results.json
Normal file
@@ -0,0 +1,9 @@
|
|||||||
|
{
|
||||||
|
"epoch": 0.8599068434252956,
|
||||||
|
"total_flos": 0.0,
|
||||||
|
"train_loss": 24.184869588216145,
|
||||||
|
"train_runtime": 7109.0777,
|
||||||
|
"train_samples": 16746,
|
||||||
|
"train_samples_per_second": 2.026,
|
||||||
|
"train_steps_per_second": 0.127
|
||||||
|
}
|
||||||
3643
trainer_state.json
Normal file
3643
trainer_state.json
Normal file
File diff suppressed because it is too large
Load Diff
3
training_args.bin
Normal file
3
training_args.bin
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:5117975d32aa6d5ee887a27ed5d586ce2b1ab4f86ce47449497c97462ed54443
|
||||||
|
size 6993
|
||||||
20
training_status.json
Normal file
20
training_status.json
Normal file
@@ -0,0 +1,20 @@
|
|||||||
|
{
|
||||||
|
"status": "completed",
|
||||||
|
"candidate_id": "inpo_b",
|
||||||
|
"method": "inpo_avg",
|
||||||
|
"seed": 42,
|
||||||
|
"gpu": 0,
|
||||||
|
"pid": 1079779,
|
||||||
|
"config": "/NHNHOME/26msit001_A/BASE/AIPR/sjkim/revision_qwen3_8b/full_iter1/fair_demo_20260715/sweep/candidates/inpo_b/config.yaml",
|
||||||
|
"dataset": "/NHNHOME/26msit001_A/BASE/AIPR/sjkim/revision_qwen3_8b/full_iter1/flagship_20260712/precomputed/avg",
|
||||||
|
"optimizer_steps": 900,
|
||||||
|
"effective_batch_size": 16,
|
||||||
|
"wandb_run_id": "098086245ba4",
|
||||||
|
"wandb_url": "https://wandb.ai/promotion-kim/mnpo/runs/098086245ba4",
|
||||||
|
"started_at": "2026-07-15T15:36:21+09:00",
|
||||||
|
"returncode": 0,
|
||||||
|
"model_complete": true,
|
||||||
|
"finite_results": true,
|
||||||
|
"measured_step": null,
|
||||||
|
"completed_at": "2026-07-15T17:35:44+09:00"
|
||||||
|
}
|
||||||
Reference in New Issue
Block a user