# Qwen3-8B LoRA config (for reference; the CLI in student_train.py accepts these as flags). base_model: Qwen/Qwen3-8B adapter: r: 16 alpha: 32 dropout: 0.05 target_modules: [q_proj, k_proj, v_proj, o_proj, gate_proj, up_proj, down_proj] training: epochs: 2 lr: 2.0e-4 batch_size: 8 grad_accum: 4 max_seq_len: 1024 warmup_ratio: 0.03 precision: bf16 packing: true load_in_4bit: true