18 lines
408 B
YAML
18 lines
408 B
YAML
# Qwen3-8B LoRA config (for reference; the CLI in student_train.py accepts these as flags).
|
|
base_model: Qwen/Qwen3-8B
|
|
adapter:
|
|
r: 16
|
|
alpha: 32
|
|
dropout: 0.05
|
|
target_modules: [q_proj, k_proj, v_proj, o_proj, gate_proj, up_proj, down_proj]
|
|
training:
|
|
epochs: 2
|
|
lr: 2.0e-4
|
|
batch_size: 8
|
|
grad_accum: 4
|
|
max_seq_len: 1024
|
|
warmup_ratio: 0.03
|
|
precision: bf16
|
|
packing: true
|
|
load_in_4bit: true
|