dolphin-2_2-yi-34b/configs/dolphin-yi-34b.yml

base_model: /workspace/models/Yi-34B-Llama
model_type: LlamaForCausalLM
tokenizer_type: AutoTokenizer
trust_remote_code: true
is_llama_derived_model: true

load_in_8bit: false
load_in_4bit: true
strict: false

datasets:
  - path: /workspace/datasets/dolphin/dolphin201.jsonl
    type: alpaca_w_system.load_open_orca_chatml
  - path: /workspace/datasets/WizardLM_evol_instruct_cleaned.jsonl
    type: sharegpt
    conversation: chatml
  - path: /workspace/datasets/not_samantha_norefusals.jsonl
    type: sharegpt
    conversation: chatml

dataset_prepared_path:
val_set_size: 0.01
output_dir: /workspace/dolphin-2.2-yi-34b

adapter: qlora
lora_model_dir:

sequence_len: 16384
sample_packing: true
pad_to_sequence_len: true

lora_r: 32
lora_alpha: 16
lora_dropout: 0.05
lora_target_modules:
lora_target_linear: true
lora_fan_in_fan_out:

lora_modules_to_save:
  - embed_tokens
  - lm_head

wandb_project: dolphin
wandb_entity:
wandb_watch:
wandb_run_id:
wandb_log_model:

gradient_accumulation_steps: 4
micro_batch_size: 1
num_epochs: 4
optimizer: paged_adamw_32bit
lr_scheduler: cosine
learning_rate: 0.0003

train_on_inputs: false
group_by_length: false
bf16: true
fp16: false
tf32: false

gradient_checkpointing: true
early_stopping_patience:
resume_from_checkpoint:
local_rank:
logging_steps: 1
xformers_attention:
flash_attention: true

warmup_steps: 100
eval_steps:
save_steps: 0.05
debug:
deepspeed: deepspeed/zero2.json
weight_decay: 0.01
fsdp:
fsdp_config:
special_tokens:
  eos_token: "<|im_end|>"
tokens:
  - "<|im_start|>"
  - "<|im_end|>"
初始化项目，由ModelHub XC社区提供模型 Model: dphn/dolphin-2_2-yi-34b Source: Original Platform 2026-05-01 06:20:21 +08:00			`base_model: /workspace/models/Yi-34B-Llama`
			`model_type: LlamaForCausalLM`
			`tokenizer_type: AutoTokenizer`
			`trust_remote_code: true`
			`is_llama_derived_model: true`

			`load_in_8bit: false`
			`load_in_4bit: true`
			`strict: false`

			`datasets:`
			`- path: /workspace/datasets/dolphin/dolphin201.jsonl`
			`type: alpaca_w_system.load_open_orca_chatml`
			`- path: /workspace/datasets/WizardLM_evol_instruct_cleaned.jsonl`
			`type: sharegpt`
			`conversation: chatml`
			`- path: /workspace/datasets/not_samantha_norefusals.jsonl`
			`type: sharegpt`
			`conversation: chatml`

			`dataset_prepared_path:`
			`val_set_size: 0.01`
			`output_dir: /workspace/dolphin-2.2-yi-34b`

			`adapter: qlora`
			`lora_model_dir:`

			`sequence_len: 16384`
			`sample_packing: true`
			`pad_to_sequence_len: true`

			`lora_r: 32`
			`lora_alpha: 16`
			`lora_dropout: 0.05`
			`lora_target_modules:`
			`lora_target_linear: true`
			`lora_fan_in_fan_out:`

			`lora_modules_to_save:`
			`- embed_tokens`
			`- lm_head`

			`wandb_project: dolphin`
			`wandb_entity:`
			`wandb_watch:`
			`wandb_run_id:`
			`wandb_log_model:`

			`gradient_accumulation_steps: 4`
			`micro_batch_size: 1`
			`num_epochs: 4`
			`optimizer: paged_adamw_32bit`
			`lr_scheduler: cosine`
			`learning_rate: 0.0003`

			`train_on_inputs: false`
			`group_by_length: false`
			`bf16: true`
			`fp16: false`
			`tf32: false`

			`gradient_checkpointing: true`
			`early_stopping_patience:`
			`resume_from_checkpoint:`
			`local_rank:`
			`logging_steps: 1`
			`xformers_attention:`
			`flash_attention: true`

			`warmup_steps: 100`
			`eval_steps:`
			`save_steps: 0.05`
			`debug:`
			`deepspeed: deepspeed/zero2.json`
			`weight_decay: 0.01`
			`fsdp:`
			`fsdp_config:`
			`special_tokens:`
			`eos_token: "<\|im_end\|>"`
			`tokens:`
			`- "<\|im_start\|>"`
			`- "<\|im_end\|>"`