67 lines
1.2 KiB
YAML
67 lines
1.2 KiB
YAML
base_model: Qwen/Qwen3-14B-Base
|
|
|
|
trust_remote_code: true
|
|
tokenizer_use_fast: true
|
|
|
|
load_in_8bit: false
|
|
load_in_4bit: false
|
|
|
|
|
|
datasets:
|
|
- path: philipperen55/datasetCPT70axolotlRandomized
|
|
split: train
|
|
data_files: datasetCPT70axolotlRandomized.jsonl
|
|
type: completion
|
|
field: text
|
|
|
|
|
|
val_set_size: 0.0001
|
|
|
|
dataset_prepared_path: prepared_cpt
|
|
output_dir: outputs_cpt
|
|
|
|
sequence_len: 2048
|
|
pad_to_sequence_len: true
|
|
sample_packing: true
|
|
eval_sample_packing: true
|
|
attn_implementation: flash_attention_2
|
|
excess_length_strategy: truncate
|
|
train_on_inputs: true
|
|
add_eos_token: true
|
|
|
|
# FULL finetune CPT
|
|
micro_batch_size: 2
|
|
gradient_accumulation_steps: 8
|
|
num_epochs: 1
|
|
|
|
|
|
optimizer: adamw_8bit
|
|
learning_rate: 4e-5
|
|
weight_decay: 0.01
|
|
lr_scheduler: constant_with_warmup
|
|
warmup_ratio: 0.01
|
|
max_grad_norm: 1.0
|
|
|
|
fp16: false
|
|
bf16: true
|
|
tf32: true
|
|
gradient_checkpointing: false
|
|
|
|
logging_steps: 50
|
|
eval_steps: 1000
|
|
save_steps: 5000
|
|
save_total_limit: 1
|
|
save_only_model: true
|
|
|
|
seed: 42
|
|
#mettre 24 si ya plus de 24 vspu, sinon mettre 16 si ya 24vcpu
|
|
dataset_num_proc: 24
|
|
|
|
# WandB
|
|
wandb_project: qwen3_14b_cpt_full_axolotl
|
|
|
|
# Hub
|
|
hub_model_id:
|
|
push_to_hub: false #mon script upload de façon plus sûre
|
|
hub_strategy: every_save
|