初始化项目,由ModelHub XC社区提供模型

Model: sugiv/qwen3-8b-tanglish
Source: Original Platform
This commit is contained in:
ModelHub XC
2026-09-08 15:34:20 +08:00
commit 5734ffb7af
437 changed files with 4525344 additions and 0 deletions

View File

@@ -0,0 +1,17 @@
# Qwen3-8B LoRA config (for reference; the CLI in student_train.py accepts these as flags).
base_model: Qwen/Qwen3-8B
adapter:
r: 16
alpha: 32
dropout: 0.05
target_modules: [q_proj, k_proj, v_proj, o_proj, gate_proj, up_proj, down_proj]
training:
epochs: 2
lr: 2.0e-4
batch_size: 8
grad_accum: 4
max_seq_len: 1024
warmup_ratio: 0.03
precision: bf16
packing: true
load_in_4bit: true