初始化项目,由ModelHub XC社区提供模型
Model: Abdine/qwen3-4b-medrect-mixed-v2 Source: Original Platform
This commit is contained in:
30
merged_from_training_config.json
Normal file
30
merged_from_training_config.json
Normal file
@@ -0,0 +1,30 @@
|
||||
{
|
||||
"train_file": "data_processed/medrect_v2/mixed_sft_train.jsonl",
|
||||
"eval_split": 0.05,
|
||||
"limit": null,
|
||||
"model_name": "Qwen/Qwen3-4B",
|
||||
"output_dir": "outputs/local_training/qwen3-4b-medrect-mixed-v2",
|
||||
"max_seq_length": 4096,
|
||||
"per_device_train_batch_size": 2,
|
||||
"gradient_accumulation_steps": 16,
|
||||
"learning_rate": 0.0001,
|
||||
"num_train_epochs": 3,
|
||||
"warmup_ratio": 0.1,
|
||||
"weight_decay": 0.1,
|
||||
"logging_steps": 10,
|
||||
"save_steps": 100,
|
||||
"eval_steps": 50,
|
||||
"bf16": true,
|
||||
"gradient_checkpointing": true,
|
||||
"lora_r": 64,
|
||||
"lora_alpha": 128,
|
||||
"lora_dropout": 0.05,
|
||||
"lora_target_modules": "all-linear",
|
||||
"early_stopping_patience": 3,
|
||||
"no_early_stopping": false,
|
||||
"wandb": true,
|
||||
"wandb_project": "medrect-mixed-sft-v2",
|
||||
"seed": 42,
|
||||
"debug_samples": 2,
|
||||
"dataloader_num_workers": 4
|
||||
}
|
||||
Reference in New Issue
Block a user