169 lines
4.0 KiB
YAML
169 lines
4.0 KiB
YAML
project:
|
|
run_name: qwen3_14b_restricted_score_ce
|
|
output_dir: runs/qwen3_14b_restricted_score_ce
|
|
seed: 42
|
|
data:
|
|
path: data/processed/response_scores.parquet
|
|
prepared_dir: null
|
|
columns:
|
|
question: question
|
|
reference_answer: ground_truth
|
|
response: response
|
|
judge_1_score: score_gpt4o
|
|
judge_2_score: score_sonnet
|
|
model_id: model_id
|
|
benchmark: benchmark
|
|
task: task
|
|
example_id: instance_id
|
|
source: source
|
|
filters:
|
|
max_disagreement: 1
|
|
max_response_tokens: 4096
|
|
on_sequence_overflow: skip
|
|
preprocessing_num_workers: 40
|
|
tokenizers_parallelism: true
|
|
token_length_batch_size: 2048
|
|
model:
|
|
model_name_or_path: Qwen/Qwen3-14B
|
|
revision: null
|
|
attn_implementation: sdpa
|
|
adapter_path: null
|
|
trust_remote_code: true
|
|
thinking_mode:
|
|
enabled: false
|
|
apply_if_supported: true
|
|
prompt:
|
|
system_prompt: 'You are an answer quality evaluator. You will receive a question,
|
|
a reference
|
|
|
|
answer, and a model response, each delimited by ### headers.
|
|
|
|
|
|
Rate the model response on the following scale:
|
|
|
|
1 - Surely incorrect: the response is wrong or completely misses the point.
|
|
|
|
2 - Likely incorrect: the response has significant errors or critical omissions.
|
|
|
|
3 - Minimally correct / sufficient: the response is partially correct or borderline.
|
|
|
|
4 - Likely correct: the response is mostly correct with minor issues.
|
|
|
|
5 - Surely correct: the response is fully correct and complete.
|
|
|
|
|
|
Rules:
|
|
|
|
- Base your rating strictly on correctness relative to the reference answer.
|
|
|
|
- Ignore writing style, verbosity, or formatting unless they affect correctness.
|
|
|
|
- If the final answer is correct but the reasoning contains substantive errors,
|
|
lower the rating according to the severity of those errors.
|
|
|
|
- Output only a single integer from 1 to 5. No explanation. No punctuation.
|
|
|
|
'
|
|
split:
|
|
mode: fixed_by_model
|
|
validation_models:
|
|
- gemini-3-flash
|
|
- DK-R1-Dist-Qwen-14B
|
|
- llama3d2-3b
|
|
train_models: auto_except_val_test
|
|
held_out_model: null
|
|
lomo_validation_fraction: 0.05
|
|
lomo_validation_max_examples: 30000
|
|
lomo_validation_seed: 42
|
|
training:
|
|
max_seq_length: 4096
|
|
load_in_4bit: true
|
|
dtype: bfloat16
|
|
objective: restricted_score_ce
|
|
loss:
|
|
type: ce_5way
|
|
lambda_binary: 0.5
|
|
class_weights: null
|
|
class_weighting: null
|
|
score_class_weights: null
|
|
lora_r: 32
|
|
lora_alpha: 64
|
|
lora_dropout: 0.0
|
|
target_modules: auto
|
|
learning_rate: 3.0e-05
|
|
num_train_epochs: 1
|
|
per_device_train_batch_size: 2
|
|
per_device_eval_batch_size: 2
|
|
gradient_accumulation_steps: 4
|
|
warmup_ratio: 0.03
|
|
lr_scheduler_type: cosine
|
|
weight_decay: 0.0
|
|
optim: adamw_8bit
|
|
packing: false
|
|
cache_tokenized_datasets: true
|
|
eval_subset_size: null
|
|
eval_subset_strategy: stratified
|
|
eval_subset_stratify_columns:
|
|
- model_id
|
|
- target_score
|
|
train_sampling_strategy: random
|
|
length_column_name: length
|
|
logging_steps: 10
|
|
eval_steps: 500
|
|
save_steps: 500
|
|
save_total_limit: 10
|
|
seed: 42
|
|
resume_from_checkpoint: null
|
|
eval_subset_seed: 42
|
|
max_grad_norm: 1.0
|
|
distributed:
|
|
enabled: true
|
|
strategy: ddp
|
|
backend: nccl
|
|
mixed_precision: bf16
|
|
gradient_checkpointing: false
|
|
find_unused_parameters: false
|
|
fsdp:
|
|
sharding_strategy: full_shard
|
|
transformer_layer_cls_to_wrap: null
|
|
activation_checkpointing: true
|
|
use_orig_params: true
|
|
deepspeed:
|
|
zero_stage: 2
|
|
offload_optimizer_device: none
|
|
offload_param_device: none
|
|
stage3_gather_16bit_weights_on_model_save: true
|
|
gradient_clipping: auto
|
|
config_overrides: {}
|
|
inference:
|
|
allowed_scores:
|
|
- '1'
|
|
- '2'
|
|
- '3'
|
|
- '4'
|
|
- '5'
|
|
binary_threshold: 3
|
|
method: restricted_continuation_logprobs_fast
|
|
generation_fallback: false
|
|
batch_size: 64
|
|
require_adapter: false
|
|
allow_base_model: true
|
|
evaluation:
|
|
length_buckets:
|
|
- 0
|
|
- 256
|
|
- 512
|
|
- 1024
|
|
- 2048
|
|
- 3072
|
|
- 4096
|
|
- 1000000000
|
|
hub:
|
|
repo_id: null
|
|
private: false
|
|
commit_message: Upload ADeLe distilled judge
|
|
local_checkpoint_dir: null
|
|
output_staging_dir: null
|
|
create_pr: false
|
|
max_shard_size: 5GB
|