初始化项目,由ModelHub XC社区提供模型
Model: sugiv/qwen3-8b-tanglish Source: Original Platform
This commit is contained in:
17
training/qwen3_8b_lora.yaml
Normal file
17
training/qwen3_8b_lora.yaml
Normal file
@@ -0,0 +1,17 @@
|
||||
# Qwen3-8B LoRA config (for reference; the CLI in student_train.py accepts these as flags).
|
||||
base_model: Qwen/Qwen3-8B
|
||||
adapter:
|
||||
r: 16
|
||||
alpha: 32
|
||||
dropout: 0.05
|
||||
target_modules: [q_proj, k_proj, v_proj, o_proj, gate_proj, up_proj, down_proj]
|
||||
training:
|
||||
epochs: 2
|
||||
lr: 2.0e-4
|
||||
batch_size: 8
|
||||
grad_accum: 4
|
||||
max_seq_len: 1024
|
||||
warmup_ratio: 0.03
|
||||
precision: bf16
|
||||
packing: true
|
||||
load_in_4bit: true
|
||||
Reference in New Issue
Block a user