初始化项目,由ModelHub XC社区提供模型

Model: promotion/ronpo-qwen3-8b-fair-dpo-s42
Source: Original Platform
This commit is contained in:
ModelHub XC
2026-07-18 16:12:11 +08:00
commit c5827a4516
18 changed files with 4248 additions and 0 deletions

36
.gitattributes vendored Normal file
View File

@@ -0,0 +1,36 @@
*.7z filter=lfs diff=lfs merge=lfs -text
*.arrow filter=lfs diff=lfs merge=lfs -text
*.bin filter=lfs diff=lfs merge=lfs -text
*.bz2 filter=lfs diff=lfs merge=lfs -text
*.ckpt filter=lfs diff=lfs merge=lfs -text
*.ftz filter=lfs diff=lfs merge=lfs -text
*.gz filter=lfs diff=lfs merge=lfs -text
*.h5 filter=lfs diff=lfs merge=lfs -text
*.joblib filter=lfs diff=lfs merge=lfs -text
*.lfs.* filter=lfs diff=lfs merge=lfs -text
*.mlmodel filter=lfs diff=lfs merge=lfs -text
*.model filter=lfs diff=lfs merge=lfs -text
*.msgpack filter=lfs diff=lfs merge=lfs -text
*.npy filter=lfs diff=lfs merge=lfs -text
*.npz filter=lfs diff=lfs merge=lfs -text
*.onnx filter=lfs diff=lfs merge=lfs -text
*.ot filter=lfs diff=lfs merge=lfs -text
*.parquet filter=lfs diff=lfs merge=lfs -text
*.pb filter=lfs diff=lfs merge=lfs -text
*.pickle filter=lfs diff=lfs merge=lfs -text
*.pkl filter=lfs diff=lfs merge=lfs -text
*.pt filter=lfs diff=lfs merge=lfs -text
*.pth filter=lfs diff=lfs merge=lfs -text
*.rar filter=lfs diff=lfs merge=lfs -text
*.safetensors filter=lfs diff=lfs merge=lfs -text
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
*.tar.* filter=lfs diff=lfs merge=lfs -text
*.tar filter=lfs diff=lfs merge=lfs -text
*.tflite filter=lfs diff=lfs merge=lfs -text
*.tgz filter=lfs diff=lfs merge=lfs -text
*.wasm filter=lfs diff=lfs merge=lfs -text
*.xz filter=lfs diff=lfs merge=lfs -text
*.zip filter=lfs diff=lfs merge=lfs -text
*.zst filter=lfs diff=lfs merge=lfs -text
*tfevents* filter=lfs diff=lfs merge=lfs -text
tokenizer.json filter=lfs diff=lfs merge=lfs -text

12
README.md Normal file
View File

@@ -0,0 +1,12 @@
---
base_model: Qwen/Qwen3-8B
library_name: transformers
tags:
- preference-optimization
- ronpo
- qwen3
---
# Qwen3-8B fair-demo checkpoint: dpo
Validation-selected candidate: `dpo_b`. Exact evaluator, selection, sweep, and training metadata are stored under `experiment/`.

9
all_results.json Normal file
View File

@@ -0,0 +1,9 @@
{
"epoch": 0.8599068434252956,
"total_flos": 0.0,
"train_loss": 0.7066660939322578,
"train_runtime": 7291.0823,
"train_samples": 16746,
"train_samples_per_second": 1.975,
"train_steps_per_second": 0.123
}

89
chat_template.jinja Normal file
View File

@@ -0,0 +1,89 @@
{%- if tools %}
{{- '<|im_start|>system\n' }}
{%- if messages[0].role == 'system' %}
{{- messages[0].content + '\n\n' }}
{%- endif %}
{{- "# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
{%- for tool in tools %}
{{- "\n" }}
{{- tool | tojson }}
{%- endfor %}
{{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
{%- else %}
{%- if messages[0].role == 'system' %}
{{- '<|im_start|>system\n' + messages[0].content + '<|im_end|>\n' }}
{%- endif %}
{%- endif %}
{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
{%- for message in messages[::-1] %}
{%- set index = (messages|length - 1) - loop.index0 %}
{%- if ns.multi_step_tool and message.role == "user" and message.content is string and not(message.content.startswith('<tool_response>') and message.content.endswith('</tool_response>')) %}
{%- set ns.multi_step_tool = false %}
{%- set ns.last_query_index = index %}
{%- endif %}
{%- endfor %}
{%- for message in messages %}
{%- if message.content is string %}
{%- set content = message.content %}
{%- else %}
{%- set content = '' %}
{%- endif %}
{%- if (message.role == "user") or (message.role == "system" and not loop.first) %}
{{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
{%- elif message.role == "assistant" %}
{%- set reasoning_content = '' %}
{%- if message.reasoning_content is string %}
{%- set reasoning_content = message.reasoning_content %}
{%- else %}
{%- if '</think>' in content %}
{%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
{%- set content = content.split('</think>')[-1].lstrip('\n') %}
{%- endif %}
{%- endif %}
{%- if loop.index0 > ns.last_query_index %}
{%- if loop.last or (not loop.last and reasoning_content) %}
{{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content.strip('\n') + '\n</think>\n\n' + content.lstrip('\n') }}
{%- else %}
{{- '<|im_start|>' + message.role + '\n' + content }}
{%- endif %}
{%- else %}
{{- '<|im_start|>' + message.role + '\n' + content }}
{%- endif %}
{%- if message.tool_calls %}
{%- for tool_call in message.tool_calls %}
{%- if (loop.first and content) or (not loop.first) %}
{{- '\n' }}
{%- endif %}
{%- if tool_call.function %}
{%- set tool_call = tool_call.function %}
{%- endif %}
{{- '<tool_call>\n{"name": "' }}
{{- tool_call.name }}
{{- '", "arguments": ' }}
{%- if tool_call.arguments is string %}
{{- tool_call.arguments }}
{%- else %}
{{- tool_call.arguments | tojson }}
{%- endif %}
{{- '}\n</tool_call>' }}
{%- endfor %}
{%- endif %}
{{- '<|im_end|>\n' }}
{%- elif message.role == "tool" %}
{%- if loop.first or (messages[loop.index0 - 1].role != "tool") %}
{{- '<|im_start|>user' }}
{%- endif %}
{{- '\n<tool_response>\n' }}
{{- content }}
{{- '\n</tool_response>' }}
{%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
{{- '<|im_end|>\n' }}
{%- endif %}
{%- endif %}
{%- endfor %}
{%- if add_generation_prompt %}
{{- '<|im_start|>assistant\n' }}
{%- if enable_thinking is defined and enable_thinking is false %}
{{- '<think>\n\n</think>\n\n' }}
{%- endif %}
{%- endif %}

71
config.json Normal file
View File

@@ -0,0 +1,71 @@
{
"architectures": [
"Qwen3ForCausalLM"
],
"attention_bias": false,
"attention_dropout": 0.0,
"bos_token_id": null,
"dtype": "bfloat16",
"eos_token_id": 151645,
"head_dim": 128,
"hidden_act": "silu",
"hidden_size": 4096,
"initializer_range": 0.02,
"intermediate_size": 12288,
"layer_types": [
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention"
],
"max_position_embeddings": 40960,
"max_window_layers": 36,
"model_type": "qwen3",
"num_attention_heads": 32,
"num_hidden_layers": 36,
"num_key_value_heads": 8,
"pad_token_id": 151643,
"rms_norm_eps": 1e-06,
"rope_parameters": {
"rope_theta": 1000000,
"rope_type": "default"
},
"sliding_window": null,
"tie_word_embeddings": false,
"transformers_version": "5.13.0",
"use_cache": true,
"use_sliding_window": false,
"vocab_size": 151936
}

58
config.yaml Normal file
View File

@@ -0,0 +1,58 @@
model_name_or_path: /NHNHOME/26msit001_A/BASE/AIPR/sjkim/revision_qwen3_8b/full_iter1/flagship_20260712/cache/huggingface/hub/models--Qwen--Qwen3-8B/snapshots/b968826d9c46dd6066d109eabc6255188de91218
torch_dtype: null
attn_implementation: sdpa
dataset_mixer:
/NHNHOME/26msit001_A/BASE/AIPR/sjkim/revision_qwen3_8b/full_iter1/flagship_20260712/precomputed/avg: 1.0
dataset_splits:
- train
- test
preprocessing_num_workers: 4
bf16: true
loss_type: dpo
eta: 0.0075
ratio: 0.3333
max_history_t: 1
history_weights:
- 1.0
dpo_beta: 0.1
simpo_beta: 2.0
simpo_gamma: 0.6
ronpo_alpha: 0.5
ronpo_tau: 0.05
ronpo_target_column: ronpo_target
reference_anchor_weight: 0.05
preference_sft_weight: 0.005
ht_target_column: ht_target
ht_target_scale: 1.0
beta: 10.0
learning_rate: 5.0e-07
lr_scheduler_type: cosine
warmup_ratio: 0.15
optim: adamw_torch
weight_decay: 0.0
max_grad_norm: 1.0
seed: 42
gradient_accumulation_steps: 16
gradient_checkpointing: true
num_train_epochs: 1
max_steps: 900
per_device_train_batch_size: 1
per_device_eval_batch_size: 1
max_length: 2048
max_prompt_length: 1800
do_eval: false
eval_strategy: 'no'
logging_steps: 5
log_level: info
generate_during_eval: false
load_best_model_at_end: false
save_strategy: steps
save_steps: 900
save_total_limit: 1
save_only_model: true
save_safetensors: true
push_to_hub: false
report_to:
- wandb
output_dir: /NHNHOME/26msit001_A/BASE/AIPR/sjkim/revision_qwen3_8b/full_iter1/fair_demo_20260715/sweep/candidates/dpo_b
run_name: qwen3-8b-fair-demo-dpo_b

View File

@@ -0,0 +1,85 @@
{
"status": "LOCKED_BEFORE_ANY_NEW_METHOD_RANKING",
"locked_at": "2026-07-15T20:20:29+09:00",
"objective_signals": [
"skywork",
"armo_safety",
"length_conciseness"
],
"objective_semantics": [
"helpfulness",
"safety",
"conciseness"
],
"primary": {
"name": "open_weight_panel_mean_prompt_worst_vs_base",
"definition": "For each prompt and objective, average win=1/tie=0.5/loss=0 over two A/B positions and two judges; take the minimum objective, then mean over prompts.",
"judges": [
{
"model": "openai/gpt-oss-120b",
"revision": "b5c939de8f754692c1647ca79fbf85e8c1e70f8a"
},
{
"model": "Qwen/Qwen3-32B",
"revision": "9216db5781bf21249d130ec9da846c4624c16137"
}
],
"decode": {
"temperature": 0.0,
"top_p": 1.0,
"max_new_tokens": 512,
"seed": 42
},
"position_swap": true,
"validation_selection_rule": "Highest eligible mean prompt-level worst panel score within each method; an exact numeric tie is broken by candidate_id lexical order."
},
"secondary": {
"normalization": "For each locked signal, divide each evaluation prompt's candidate-minus-base raw delta by the population SD of the diagnostic base and matched-control scores; take the prompt-level minimum across objectives, then the mean. No per-prompt min-max.",
"diagnostic_control_scale": {
"skywork": 22.807301785782975,
"armo_safety": 0.3098083353203591,
"length_conciseness": 1.5891969646897472
},
"reward_signal_provenance": {
"skywork": {
"model": "Skywork/Skywork-Reward-V2-Llama-3.1-8B",
"revision": "cba2f842f3f1af2f1b2f0d35e794d789976390c5",
"semantics": "helpfulness"
},
"armo_safety": {
"model": "RLHFlow/ArmoRM-Llama3-8B-v0.1",
"revision": "eb2676d20da2f2d41082289d23c59b9f7427f955",
"head": "beavertails-is_safe",
"transform": "identity"
},
"length_conciseness": {
"deterministic": "-log1p(response_word_count)",
"tokenization": "Python str.split whitespace words"
}
},
"report_raw_paired_deltas": true
},
"bootstrap": {
"paired_prompt_resamples": 2000,
"seed": 42,
"interval": "percentile_95"
},
"power": {
"target_absolute_effect": 0.05,
"paired_sd": 0.13831629563357775,
"required_prompts_80pct_power": 61,
"planned_fresh_test_prompts": 1024,
"alpha_two_sided": 0.05,
"power": 0.8
},
"fresh_test_source": {
"dataset": "HuggingFaceH4/ultrachat_200k",
"revision": "8049631c405ae6576f93f445c6b8166f76f5505a",
"split": "test_sft"
},
"diagnostic_summary_sha256": "e365ed814926ac5c17777e0fc65d0299a45b81bd390542889bbd55b5453d4ffe",
"judge_diagnostic_lock_sha256": "fce2d97806053cabdf3d358806f4baa80df7f89d71d8c1c298883327d4e02635",
"prereg_sha256": "71cbd1ac17c3e807f2ece1cb0a82374bbe74edb654c2845b97c3222ed680f0d0",
"method_ranking_computed": false,
"spent_sealed_split_touched": false
}

View File

@@ -0,0 +1,99 @@
{
"status": "VALIDATION_SELECTION_LOCKED_BEFORE_FRESH_TEST",
"locked_at": "2026-07-15T20:37:11+09:00",
"selection_metric": "open_weight_panel_mean_prompt_worst_vs_base",
"tie_break": "candidate_id lexical order",
"selected_by_method": {
"dpo": {
"candidate_id": "dpo_b",
"validation_primary": 0.40625,
"validation_primary_ci95": [
0.3837890625,
0.427734375
],
"eligible_candidates": [
"dpo_a",
"dpo_b"
]
},
"ht_mnpo_helpfulness": {
"candidate_id": "ht_help_b",
"validation_primary": 0.4072265625,
"validation_primary_ci95": [
0.3828125,
0.4296875
],
"eligible_candidates": [
"ht_help_b"
]
},
"ht_mnpo_safety": {
"candidate_id": "ht_safety_a",
"validation_primary": 0.4150390625,
"validation_primary_ci95": [
0.3925537109375,
0.4384765625
],
"eligible_candidates": [
"ht_safety_a",
"ht_safety_b"
]
},
"inpo_avg": {
"candidate_id": "inpo_b",
"validation_primary": 0.40625,
"validation_primary_ci95": [
0.3818359375,
0.427734375
],
"eligible_candidates": [
"inpo_b"
]
},
"ronpo_full_expect": {
"candidate_id": "ronpo_full_a",
"validation_primary": 0.4033203125,
"validation_primary_ci95": [
0.37890625,
0.42580566406249987
],
"eligible_candidates": [
"ronpo_full_a",
"ronpo_full_b"
]
},
"ronpo_k_only": {
"candidate_id": "ronpo_top_a",
"validation_primary": 0.412109375,
"validation_primary_ci95": [
0.3876953125,
0.43557128906249987
],
"eligible_candidates": [
"ronpo_top_a"
]
},
"sppo_avg": {
"candidate_id": "sppo_b",
"validation_primary": 0.4111328125,
"validation_primary_ci95": [
0.3866943359375,
0.4345703125
],
"eligible_candidates": [
"sppo_b"
]
}
},
"selected_ronpo_overall": "ronpo_top_a",
"failed_methods": [
"ht_mnpo_conciseness",
"ipo",
"simpo"
],
"panel_summary_sha256": "8850f23e16cadf56a99e3d562d224ce5337ab36fa588d8a843382853e6ad17bd",
"evaluator_lock_sha256": "ab7b12c84932ed57acfec7edbcbc1ff1df5e9acd203869b1458343adf6e5b2c8",
"grid_sha256": "abfd982ef10c3bc71364752241150286ebf06be5e97f087596adf80e55d459e0",
"fresh_test_opened": false,
"spent_sealed_split_touched": false
}

View File

@@ -0,0 +1,46 @@
{
"schema_version": 1,
"status": "frozen_before_training_and_validation_ranking",
"frozen_at_kst": "2026-07-15T09:01:45+09:00",
"seed": 42,
"common": {
"optimizer_steps": 900,
"effective_batch_size": 16,
"per_device_train_batch_size": 1,
"gradient_accumulation_steps": 16,
"gradient_checkpointing": true,
"bf16": true,
"attn_implementation": "sdpa",
"cudnn_sdpa": false,
"lr_scheduler_type": "cosine",
"max_length": 2048,
"max_prompt_length": 1800,
"save_total_limit": 1,
"wandb_entity": "promotion-kim",
"wandb_project": "mnpo",
"wandb_group": "qwen3-8b-fair-demo-20260715"
},
"budget_rule": "Exactly two configs per reported method; no extension after validation rankings are visible.",
"candidates": [
{"id":"ronpo_full_a","method":"ronpo_full_expect","dataset":"ronpo_full_expect_kall","learning_rate":1e-7,"warmup_ratio":0.15,"ronpo_alpha":0.25,"ronpo_tau":0.10,"reference_anchor_weight":0.075,"preference_sft_weight":0.0075,"expected_support_k":0},
{"id":"ronpo_full_b","method":"ronpo_full_expect","dataset":"ronpo_full_expect_k6","learning_rate":2.5e-7,"warmup_ratio":0.20,"ronpo_alpha":0.35,"ronpo_tau":0.05,"reference_anchor_weight":0.10,"preference_sft_weight":0.01,"expected_support_k":6},
{"id":"ronpo_top_a","method":"ronpo_k_only","dataset":"ronpo_k_only_k1","learning_rate":5e-8,"warmup_ratio":0.20,"ronpo_alpha":0.15,"ronpo_tau":0.10,"reference_anchor_weight":0.10,"preference_sft_weight":0.01,"expected_support_k":1},
{"id":"ronpo_top_b","method":"ronpo_k_only","dataset":"ronpo_k_only_k2","learning_rate":1e-7,"warmup_ratio":0.15,"ronpo_alpha":0.25,"ronpo_tau":0.10,"reference_anchor_weight":0.075,"preference_sft_weight":0.0075,"expected_support_k":2},
{"id":"dpo_a","method":"dpo","dataset":"avg","learning_rate":2.5e-7,"warmup_ratio":0.10,"dpo_beta":0.05,"reference_anchor_weight":0.02,"preference_sft_weight":0.002},
{"id":"dpo_b","method":"dpo","dataset":"avg","learning_rate":5e-7,"warmup_ratio":0.15,"dpo_beta":0.10,"reference_anchor_weight":0.05,"preference_sft_weight":0.005},
{"id":"ipo_a","method":"ipo","dataset":"avg","learning_rate":2.5e-7,"warmup_ratio":0.15,"dpo_beta":0.20,"reference_anchor_weight":0.05,"preference_sft_weight":0.005},
{"id":"ipo_b","method":"ipo","dataset":"avg","learning_rate":5e-7,"warmup_ratio":0.20,"dpo_beta":0.50,"reference_anchor_weight":0.075,"preference_sft_weight":0.0075},
{"id":"simpo_a","method":"simpo","dataset":"avg","learning_rate":2.5e-7,"warmup_ratio":0.10,"simpo_beta":1.0,"simpo_gamma":0.3,"reference_anchor_weight":0.02,"preference_sft_weight":0.002},
{"id":"simpo_b","method":"simpo","dataset":"avg","learning_rate":5e-7,"warmup_ratio":0.15,"simpo_beta":2.0,"simpo_gamma":0.6,"reference_anchor_weight":0.05,"preference_sft_weight":0.005},
{"id":"sppo_a","method":"sppo_avg","dataset":"avg","learning_rate":2.5e-7,"warmup_ratio":0.15,"eta":0.005,"reference_anchor_weight":0.05,"preference_sft_weight":0.005},
{"id":"sppo_b","method":"sppo_avg","dataset":"avg","learning_rate":5e-7,"warmup_ratio":0.20,"eta":0.010,"reference_anchor_weight":0.075,"preference_sft_weight":0.0075},
{"id":"inpo_a","method":"inpo_avg","dataset":"avg","learning_rate":2.5e-7,"warmup_ratio":0.15,"eta":0.050,"ratio":0.3333,"reference_anchor_weight":0.05,"preference_sft_weight":0.005},
{"id":"inpo_b","method":"inpo_avg","dataset":"avg","learning_rate":5e-7,"warmup_ratio":0.20,"eta":0.100,"ratio":0.3333,"reference_anchor_weight":0.075,"preference_sft_weight":0.0075},
{"id":"ht_help_a","method":"ht_mnpo_helpfulness","dataset":"ht_mnpo_helpfulness","learning_rate":2.5e-7,"warmup_ratio":0.10,"eta":0.005,"reference_anchor_weight":0.02,"preference_sft_weight":0.002},
{"id":"ht_help_b","method":"ht_mnpo_helpfulness","dataset":"ht_mnpo_helpfulness","learning_rate":5e-7,"warmup_ratio":0.15,"eta":0.010,"reference_anchor_weight":0.05,"preference_sft_weight":0.005},
{"id":"ht_safety_a","method":"ht_mnpo_safety","dataset":"ht_mnpo_safety","learning_rate":2.5e-7,"warmup_ratio":0.10,"eta":0.005,"reference_anchor_weight":0.02,"preference_sft_weight":0.002},
{"id":"ht_safety_b","method":"ht_mnpo_safety","dataset":"ht_mnpo_safety","learning_rate":5e-7,"warmup_ratio":0.15,"eta":0.010,"reference_anchor_weight":0.05,"preference_sft_weight":0.005},
{"id":"ht_concise_a","method":"ht_mnpo_conciseness","dataset":"ht_mnpo_conciseness","learning_rate":2.5e-7,"warmup_ratio":0.10,"eta":0.005,"reference_anchor_weight":0.02,"preference_sft_weight":0.002},
{"id":"ht_concise_b","method":"ht_mnpo_conciseness","dataset":"ht_mnpo_conciseness","learning_rate":5e-7,"warmup_ratio":0.15,"eta":0.010,"reference_anchor_weight":0.05,"preference_sft_weight":0.005}
]
}

View File

@@ -0,0 +1,20 @@
{
"status": "completed",
"candidate_id": "dpo_b",
"method": "dpo",
"seed": 42,
"gpu": 0,
"pid": 423793,
"config": "/NHNHOME/26msit001_A/BASE/AIPR/sjkim/revision_qwen3_8b/full_iter1/fair_demo_20260715/sweep/candidates/dpo_b/config.yaml",
"dataset": "/NHNHOME/26msit001_A/BASE/AIPR/sjkim/revision_qwen3_8b/full_iter1/flagship_20260712/precomputed/avg",
"optimizer_steps": 900,
"effective_batch_size": 16,
"wandb_run_id": "bce8f73e973a",
"wandb_url": "https://wandb.ai/promotion-kim/mnpo/runs/bce8f73e973a",
"started_at": "2026-07-15T11:34:23+09:00",
"returncode": 0,
"model_complete": true,
"finite_results": true,
"measured_step": null,
"completed_at": "2026-07-15T13:36:52+09:00"
}

12
generation_config.json Normal file
View File

@@ -0,0 +1,12 @@
{
"do_sample": true,
"eos_token_id": [
151645,
151643
],
"pad_token_id": 151643,
"temperature": 0.6,
"top_k": 20,
"top_p": 0.95,
"transformers_version": "5.13.0"
}

3
model.safetensors Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:065758f41205823cfe84fa5d700950d002bbc87aad92fa59581e52db7bdb1044
size 16381517208

3
tokenizer.json Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:be75606093db2094d7cd20f3c2f385c212750648bd6ea4fb2bf507a6a4c55506
size 11422650

30
tokenizer_config.json Normal file
View File

@@ -0,0 +1,30 @@
{
"add_prefix_space": false,
"backend": "tokenizers",
"bos_token": null,
"clean_up_tokenization_spaces": false,
"eos_token": "<|im_end|>",
"errors": "replace",
"extra_special_tokens": [
"<|im_start|>",
"<|im_end|>",
"<|object_ref_start|>",
"<|object_ref_end|>",
"<|box_start|>",
"<|box_end|>",
"<|quad_start|>",
"<|quad_end|>",
"<|vision_start|>",
"<|vision_end|>",
"<|vision_pad|>",
"<|image_pad|>",
"<|video_pad|>"
],
"is_local": true,
"local_files_only": true,
"model_max_length": 2048,
"pad_token": "<|endoftext|>",
"split_special_tokens": false,
"tokenizer_class": "Qwen2Tokenizer",
"unk_token": null
}

9
train_results.json Normal file
View File

@@ -0,0 +1,9 @@
{
"epoch": 0.8599068434252956,
"total_flos": 0.0,
"train_loss": 0.7066660939322578,
"train_runtime": 7291.0823,
"train_samples": 16746,
"train_samples_per_second": 1.975,
"train_steps_per_second": 0.123
}

3643
trainer_state.json Normal file

File diff suppressed because it is too large Load Diff

3
training_args.bin Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:14efa4061992d64e8388448d5d8e71f0820fe2e786143d9fa23f78658612645f
size 6993

20
training_status.json Normal file
View File

@@ -0,0 +1,20 @@
{
"status": "completed",
"candidate_id": "dpo_b",
"method": "dpo",
"seed": 42,
"gpu": 0,
"pid": 423793,
"config": "/NHNHOME/26msit001_A/BASE/AIPR/sjkim/revision_qwen3_8b/full_iter1/fair_demo_20260715/sweep/candidates/dpo_b/config.yaml",
"dataset": "/NHNHOME/26msit001_A/BASE/AIPR/sjkim/revision_qwen3_8b/full_iter1/flagship_20260712/precomputed/avg",
"optimizer_steps": 900,
"effective_batch_size": 16,
"wandb_run_id": "bce8f73e973a",
"wandb_url": "https://wandb.ai/promotion-kim/mnpo/runs/bce8f73e973a",
"started_at": "2026-07-15T11:34:23+09:00",
"returncode": 0,
"model_complete": true,
"finite_results": true,
"measured_step": null,
"completed_at": "2026-07-15T13:36:52+09:00"
}