初始化项目，由ModelHub XC社区提供模型

Model: the-jb/tofu_Llama-3.2-3B-Instruct_forget10_GradDiff Source: Original Platform
2026-05-12 20:44:39 +08:00
commit 7f97a0382d
23 changed files with 42747 additions and 0 deletions
--- a/.gitattributes
+++ b/.gitattributes
@@ -0,0 +1,36 @@
 *.7z filter=lfs diff=lfs merge=lfs -text
 *.arrow filter=lfs diff=lfs merge=lfs -text
 *.bin filter=lfs diff=lfs merge=lfs -text
 *.bz2 filter=lfs diff=lfs merge=lfs -text
 *.ckpt filter=lfs diff=lfs merge=lfs -text
 *.ftz filter=lfs diff=lfs merge=lfs -text
 *.gz filter=lfs diff=lfs merge=lfs -text
 *.h5 filter=lfs diff=lfs merge=lfs -text
 *.joblib filter=lfs diff=lfs merge=lfs -text
 *.lfs.* filter=lfs diff=lfs merge=lfs -text
 *.mlmodel filter=lfs diff=lfs merge=lfs -text
 *.model filter=lfs diff=lfs merge=lfs -text
 *.msgpack filter=lfs diff=lfs merge=lfs -text
 *.npy filter=lfs diff=lfs merge=lfs -text
 *.npz filter=lfs diff=lfs merge=lfs -text
 *.onnx filter=lfs diff=lfs merge=lfs -text
 *.ot filter=lfs diff=lfs merge=lfs -text
 *.parquet filter=lfs diff=lfs merge=lfs -text
 *.pb filter=lfs diff=lfs merge=lfs -text
 *.pickle filter=lfs diff=lfs merge=lfs -text
 *.pkl filter=lfs diff=lfs merge=lfs -text
 *.pt filter=lfs diff=lfs merge=lfs -text
 *.pth filter=lfs diff=lfs merge=lfs -text
 *.rar filter=lfs diff=lfs merge=lfs -text
 *.safetensors filter=lfs diff=lfs merge=lfs -text
 saved_model/**/* filter=lfs diff=lfs merge=lfs -text
 *.tar.* filter=lfs diff=lfs merge=lfs -text
 *.tar filter=lfs diff=lfs merge=lfs -text
 *.tflite filter=lfs diff=lfs merge=lfs -text
 *.tgz filter=lfs diff=lfs merge=lfs -text
 *.wasm filter=lfs diff=lfs merge=lfs -text
 *.xz filter=lfs diff=lfs merge=lfs -text
 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text
 tokenizer.json filter=lfs diff=lfs merge=lfs -text
--- a/.hydra/config.yaml
+++ b/.hydra/config.yaml
@@ -0,0 +1,609 @@
 model:
  model_args:
    pretrained_model_name_or_path: open-unlearning/tofu_Llama-3.2-3B-Instruct_full
    attn_implementation: flash_attention_2
    torch_dtype: bfloat16
  tokenizer_args:
    pretrained_model_name_or_path: meta-llama/Llama-3.2-3B-Instruct
  template_args:
    apply_chat_template: true
    system_prompt: You are a helpful assistant.
    system_prompt_with_special_tokens: '<|begin_of_text|><|start_header_id|>system<|end_header_id|>
      You are a helpful assistant.<|eot_id|>'
    user_start_tag: '<|start_header_id|>user<|end_header_id|>
      '
    user_end_tag: <|eot_id|>
    asst_start_tag: '<|start_header_id|>assistant<|end_header_id|>
      '
    asst_end_tag: <|eot_id|>
    date_string: 10 Apr 2025
 trainer:
  handler: GradDiff
  args:
    per_device_train_batch_size: 4
    per_device_eval_batch_size: 16
    gradient_accumulation_steps: 4
    learning_rate: 1.0e-05
    bf16: true
    bf16_full_eval: true
    logging_steps: 5
    output_dir: ${paths.output_dir}
    logging_dir: ${trainer.args.output_dir}/logs
    report_to: tensorboard
    ddp_find_unused_parameters: true
    gradient_checkpointing: true
    optim: paged_adamw_32bit
    save_strategy: 'no'
    save_only_model: true
    weight_decay: 0.01
    do_train: true
    do_eval: true
    eval_on_start: true
    eval_strategy: epoch
    num_train_epochs: 10
    seed: 0
    warmup_epochs: 1.0
    remove_unused_columns: false
  method_args:
    gamma: 1.0
    alpha: 1.0
    retain_loss_type: NLL
 data:
  forget:
    TOFU_QA_forget:
      handler: QADataset
      args:
        hf_args:
          name: ${forget_split}
          split: train
          path: locuslab/TOFU
        question_key: question
        answer_key: answer
        max_length: 512
  retain:
    TOFU_QA_retain:
      handler: QADataset
      args:
        hf_args:
          name: ${retain_split}_wo_test
          split: train
          path: locuslab/TOFU
        question_key: question
        answer_key: answer
        max_length: 512
  anchor: forget
 collator:
  DataCollatorForSupervisedDataset:
    handler: DataCollatorForSupervisedDataset
    args:
      padding_side: right
 eval:
  tofu:
    metrics:
      forget_quality:
        pre_compute:
          forget_truth_ratio:
            pre_compute:
              forget_Q_A_PARA_Prob:
                datasets:
                  TOFU_QA_forget_para:
                    handler: QADataset
                    args:
                      hf_args:
                        name: ${eval.tofu.forget_split}_perturbed
                        split: train
                        path: locuslab/TOFU
                      question_key: question
                      answer_key: paraphrased_answer
                      max_length: 512
                collators:
                  DataCollatorForSupervisedDataset:
                    handler: DataCollatorForSupervisedDataset
                    args:
                      padding_side: right
                      index: index
                handler: probability
                batch_size: 32
                access_key: correct
              forget_Q_A_PERT_Prob:
                datasets:
                  TOFU_QA_forget_pert:
                    handler: QADataset
                    args:
                      hf_args:
                        name: ${eval.tofu.forget_split}_perturbed
                        split: train
                        path: locuslab/TOFU
                      question_key: question
                      answer_key: perturbed_answer
                      max_length: 512
                collators:
                  DataCollatorForSupervisedDataset:
                    handler: DataCollatorForSupervisedDataset
                    args:
                      padding_side: right
                      index: index
                handler: probability
                batch_size: 32
                access_key: wrong
            handler: truth_ratio
            aggregator: closer_to_1_better
            access_key: forget
        reference_logs:
          retain_model_logs:
            path: ${eval.tofu.retain_logs_path}
            include:
              forget_truth_ratio:
                access_key: retain
        handler: ks_test
      forget_Q_A_Prob:
        datasets:
          TOFU_QA_forget:
            handler: QADataset
            args:
              hf_args:
                name: ${eval.tofu.forget_split}
                split: train
                path: locuslab/TOFU
              question_key: question
              answer_key: answer
              max_length: 512
        collators:
          DataCollatorForSupervisedDataset:
            handler: DataCollatorForSupervisedDataset
            args:
              padding_side: right
              index: index
        handler: probability
        batch_size: 32
      forget_Q_A_ROUGE:
        datasets:
          TOFU_QA_forget:
            handler: QADataset
            args:
              hf_args:
                name: ${eval.tofu.forget_split}
                split: train
                path: locuslab/TOFU
              question_key: question
              answer_key: answer
              max_length: 512
              predict_with_generate: true
        collators:
          DataCollatorForSupervisedDataset:
            handler: DataCollatorForSupervisedDataset
            args:
              padding_side: left
              index: index
        generation_args:
          do_sample: false
          top_p: null
          temperature: null
          max_new_tokens: 200
          use_cache: true
        handler: rouge
        rouge_type: rougeL_recall
        batch_size: 32
      model_utility:
        pre_compute:
          retain_Q_A_Prob:
            datasets:
              TOFU_QA_retain_eval:
                handler: QADataset
                args:
                  hf_args:
                    name: retain_perturbed
                    split: train
                    path: locuslab/TOFU
                  question_key: question
                  answer_key: answer
                  max_length: 512
            collators:
              DataCollatorForSupervisedDataset:
                handler: DataCollatorForSupervisedDataset
                args:
                  padding_side: right
                  index: index
            handler: probability
            batch_size: 32
          retain_Q_A_ROUGE:
            datasets:
              TOFU_QA_retain_eval:
                handler: QADataset
                args:
                  hf_args:
                    name: retain_perturbed
                    split: train
                    path: locuslab/TOFU
                  question_key: question
                  answer_key: answer
                  max_length: 512
                  predict_with_generate: true
            collators:
              DataCollatorForSupervisedDataset:
                handler: DataCollatorForSupervisedDataset
                args:
                  padding_side: left
                  index: index
            generation_args:
              do_sample: false
              top_p: null
              temperature: null
              max_new_tokens: 200
              use_cache: true
            handler: rouge
            rouge_type: rougeL_recall
            batch_size: 32
          retain_Truth_Ratio:
            pre_compute:
              retain_Q_A_PARA_Prob:
                datasets:
                  TOFU_QA_retain_para:
                    handler: QADataset
                    args:
                      hf_args:
                        name: retain_perturbed
                        split: train
                        path: locuslab/TOFU
                      question_key: question
                      answer_key: paraphrased_answer
                      max_length: 512
                collators:
                  DataCollatorForSupervisedDataset:
                    handler: DataCollatorForSupervisedDataset
                    args:
                      padding_side: right
                      index: index
                handler: probability
                batch_size: 32
                access_key: correct
              retain_Q_A_PERT_Prob:
                datasets:
                  TOFU_QA_retain_pert:
                    handler: QADataset
                    args:
                      hf_args:
                        name: retain_perturbed
                        split: train
                        path: locuslab/TOFU
                      question_key: question
                      answer_key: perturbed_answer
                      max_length: 512
                collators:
                  DataCollatorForSupervisedDataset:
                    handler: DataCollatorForSupervisedDataset
                    args:
                      padding_side: right
                      index: index
                handler: probability
                batch_size: 32
                access_key: wrong
            handler: truth_ratio
            aggregator: true_better
          ra_Q_A_Prob_normalised:
            pre_compute:
              ra_Q_A_Prob:
                datasets:
                  TOFU_QA_ra:
                    handler: QADataset
                    args:
                      hf_args:
                        name: real_authors_perturbed
                        split: train
                        path: locuslab/TOFU
                      question_key: question
                      answer_key: answer
                      max_length: 512
                collators:
                  DataCollatorForSupervisedDataset:
                    handler: DataCollatorForSupervisedDataset
                    args:
                      padding_side: right
                      index: index
                handler: probability
                batch_size: 32
                access_key: correct
              ra_Q_A_PERT_Prob:
                datasets:
                  TOFU_QA_ra_pert:
                    handler: QADataset
                    args:
                      hf_args:
                        name: real_authors_perturbed
                        split: train
                        path: locuslab/TOFU
                      question_key: question
                      answer_key: perturbed_answer
                      max_length: 512
                collators:
                  DataCollatorForSupervisedDataset:
                    handler: DataCollatorForSupervisedDataset
                    args:
                      padding_side: right
                      index: index
                handler: probability
                batch_size: 32
                access_key: wrong
            handler: probability_w_options
          ra_Q_A_ROUGE:
            datasets:
              TOFU_QA_ra:
                handler: QADataset
                args:
                  hf_args:
                    name: real_authors_perturbed
                    split: train
                    path: locuslab/TOFU
                  question_key: question
                  answer_key: answer
                  max_length: 512
                  predict_with_generate: true
            collators:
              DataCollatorForSupervisedDataset:
                handler: DataCollatorForSupervisedDataset
                args:
                  padding_side: left
                  index: index
            generation_args:
              do_sample: false
              top_p: null
              temperature: null
              max_new_tokens: 200
              use_cache: true
            handler: rouge
            rouge_type: rougeL_recall
            batch_size: 32
          ra_Truth_Ratio:
            pre_compute:
              ra_Q_A_Prob:
                datasets:
                  TOFU_QA_ra:
                    handler: QADataset
                    args:
                      hf_args:
                        name: real_authors_perturbed
                        split: train
                        path: locuslab/TOFU
                      question_key: question
                      answer_key: answer
                      max_length: 512
                collators:
                  DataCollatorForSupervisedDataset:
                    handler: DataCollatorForSupervisedDataset
                    args:
                      padding_side: right
                      index: index
                handler: probability
                batch_size: 32
                access_key: correct
              ra_Q_A_PERT_Prob:
                datasets:
                  TOFU_QA_ra_pert:
                    handler: QADataset
                    args:
                      hf_args:
                        name: real_authors_perturbed
                        split: train
                        path: locuslab/TOFU
                      question_key: question
                      answer_key: perturbed_answer
                      max_length: 512
                collators:
                  DataCollatorForSupervisedDataset:
                    handler: DataCollatorForSupervisedDataset
                    args:
                      padding_side: right
                      index: index
                handler: probability
                batch_size: 32
                access_key: wrong
            handler: truth_ratio
            aggregator: true_better
          wf_Q_A_Prob_normalised:
            pre_compute:
              wf_Q_A_Prob:
                datasets:
                  TOFU_QA_wf:
                    handler: QADataset
                    args:
                      hf_args:
                        name: world_facts_perturbed
                        split: train
                        path: locuslab/TOFU
                      question_key: question
                      answer_key: answer
                      max_length: 512
                collators:
                  DataCollatorForSupervisedDataset:
                    handler: DataCollatorForSupervisedDataset
                    args:
                      padding_side: right
                      index: index
                handler: probability
                batch_size: 32
                access_key: correct
              wf_Q_A_PERT_Prob:
                datasets:
                  TOFU_QA_wf_pert:
                    handler: QADataset
                    args:
                      hf_args:
                        name: world_facts_perturbed
                        split: train
                        path: locuslab/TOFU
                      question_key: question
                      answer_key: perturbed_answer
                      max_length: 512
                collators:
                  DataCollatorForSupervisedDataset:
                    handler: DataCollatorForSupervisedDataset
                    args:
                      padding_side: right
                      index: index
                handler: probability
                batch_size: 32
                access_key: wrong
            handler: probability_w_options
          wf_Q_A_ROUGE:
            datasets:
              TOFU_QA_wf:
                handler: QADataset
                args:
                  hf_args:
                    name: world_facts_perturbed
                    split: train
                    path: locuslab/TOFU
                  question_key: question
                  answer_key: answer
                  max_length: 512
                  predict_with_generate: true
            collators:
              DataCollatorForSupervisedDataset:
                handler: DataCollatorForSupervisedDataset
                args:
                  padding_side: left
                  index: index
            generation_args:
              do_sample: false
              top_p: null
              temperature: null
              max_new_tokens: 200
              use_cache: true
            handler: rouge
            rouge_type: rougeL_recall
            batch_size: 32
          wf_Truth_Ratio:
            pre_compute:
              wf_Q_A_Prob:
                datasets:
                  TOFU_QA_wf:
                    handler: QADataset
                    args:
                      hf_args:
                        name: world_facts_perturbed
                        split: train
                        path: locuslab/TOFU
                      question_key: question
                      answer_key: answer
                      max_length: 512
                collators:
                  DataCollatorForSupervisedDataset:
                    handler: DataCollatorForSupervisedDataset
                    args:
                      padding_side: right
                      index: index
                handler: probability
                batch_size: 32
                access_key: correct
              wf_Q_A_PERT_Prob:
                datasets:
                  TOFU_QA_wf_pert:
                    handler: QADataset
                    args:
                      hf_args:
                        name: world_facts_perturbed
                        split: train
                        path: locuslab/TOFU
                      question_key: question
                      answer_key: perturbed_answer
                      max_length: 512
                collators:
                  DataCollatorForSupervisedDataset:
                    handler: DataCollatorForSupervisedDataset
                    args:
                      padding_side: right
                      index: index
                handler: probability
                batch_size: 32
                access_key: wrong
            handler: truth_ratio
            aggregator: true_better
        handler: hm_aggregate
      privleak:
        pre_compute:
          mia_min_k:
            datasets:
              TOFU_QA_forget:
                access_key: forget
                handler: QADataset
                args:
                  hf_args:
                    name: ${eval.tofu.forget_split}
                    split: train
                    path: locuslab/TOFU
                  question_key: question
                  answer_key: answer
                  max_length: 512
              TOFU_QA_holdout:
                access_key: holdout
                handler: QADataset
                args:
                  hf_args:
                    name: ${eval.tofu.holdout_split}
                    path: locuslab/TOFU
                    split: train
                  question_key: question
                  answer_key: answer
                  max_length: 512
            collators:
              DataCollatorForSupervisedDataset:
                handler: DataCollatorForSupervisedDataset
                args:
                  padding_side: right
                  index: index
            batch_size: 32
            handler: mia_min_k
            k: 0.4
            access_key: forget
        reference_logs:
          retain_model_logs:
            path: ${eval.tofu.retain_logs_path}
            include:
              mia_min_k:
                access_key: retain
        handler: privleak
        ref_value: 0.5
      extraction_strength:
        datasets:
          TOFU_QA_forget:
            handler: QADataset
            args:
              hf_args:
                name: ${eval.tofu.forget_split}
                split: train
                path: locuslab/TOFU
              question_key: question
              answer_key: answer
              max_length: 512
        collators:
          DataCollatorForSupervisedDataset:
            handler: DataCollatorForSupervisedDataset
            args:
              padding_side: right
              index: index
        handler: extraction_strength
        batch_size: 32
    handler: TOFUEvaluator
    output_dir: ${paths.output_dir}
    overwrite: true
    forget_split: ${forget_split}
    holdout_split: ${holdout_split}
    retain_logs_path: ${retain_logs_path}
 paths:
  root_dir: .
  data_dir: ${paths.root_dir}/data/
  datasets: ${paths.root_dir}/configs/data/datasets
  output_dir: ${paths.root_dir}/saves/${mode}/${task_name}
  work_dir: ${hydra:runtime.cwd}
 forget_split: forget10
 retain_split: retain90
 holdout_split: holdout10
 retain_logs_path: saves/eval/tofu_Llama-3.2-3B-Instruct_retain90/TOFU_EVAL.json
 task_name: tofu_Llama-3.2-3B-Instruct_forget10_GradDiff
 mode: unlearn
--- a/.hydra/hydra.yaml
+++ b/.hydra/hydra.yaml
@@ -0,0 +1,279 @@
 hydra:
  run:
    dir: ${paths.output_dir}
  sweep:
    dir: multirun/${now:%Y-%m-%d}/${now:%H-%M-%S}
    subdir: ${hydra.job.num}
  launcher:
    _target_: hydra._internal.core_plugins.basic_launcher.BasicLauncher
  sweeper:
    _target_: hydra._internal.core_plugins.basic_sweeper.BasicSweeper
    max_batch_size: null
    params: null
  help:
    app_name: ${hydra.job.name}
    header: '${hydra.help.app_name} is powered by Hydra.
      '
    footer: 'Powered by Hydra (https://hydra.cc)
      Use --hydra-help to view Hydra specific help
      '
    template: '${hydra.help.header}
      == Configuration groups ==
      Compose your configuration from those groups (group=option)
      $APP_CONFIG_GROUPS
      == Config ==
      Override anything in the config (foo.bar=value)
      $CONFIG
      ${hydra.help.footer}
      '
  hydra_help:
    template: 'Hydra (${hydra.runtime.version})
      See https://hydra.cc for more info.
      == Flags ==
      $FLAGS_HELP
      == Configuration groups ==
      Compose your configuration from those groups (For example, append hydra/job_logging=disabled
      to command line)
      $HYDRA_CONFIG_GROUPS
      Use ''--cfg hydra'' to Show the Hydra config.
      '
    hydra_help: ???
  hydra_logging:
    version: 1
    formatters:
      colorlog:
        (): colorlog.ColoredFormatter
        format: '[%(cyan)s%(asctime)s%(reset)s][%(purple)sHYDRA%(reset)s] %(message)s'
    handlers:
      console:
        class: logging.StreamHandler
        formatter: colorlog
        stream: ext://sys.stdout
    root:
      level: INFO
      handlers:
      - console
    disable_existing_loggers: false
  job_logging:
    version: 1
    formatters:
      simple:
        format: '[%(asctime)s][%(name)s][%(levelname)s] - %(message)s'
      colorlog:
        (): colorlog.ColoredFormatter
        format: '[%(cyan)s%(asctime)s%(reset)s][%(blue)s%(name)s%(reset)s][%(log_color)s%(levelname)s%(reset)s]
          - %(message)s'
        log_colors:
          DEBUG: purple
          INFO: green
          WARNING: yellow
          ERROR: red
          CRITICAL: red
    handlers:
      console:
        class: logging.StreamHandler
        formatter: colorlog
        stream: ext://sys.stdout
      file:
        class: logging.FileHandler
        formatter: simple
        filename: ${hydra.runtime.output_dir}/${trainer.handler}.log
    root:
      level: INFO
      handlers:
      - console
      - file
    disable_existing_loggers: false
  env: {}
  mode: RUN
  searchpath: []
  callbacks: {}
  output_subdir: .hydra
  overrides:
    hydra:
    - hydra.mode=RUN
    task:
    - experiment=unlearn/tofu/default.yaml
    - trainer=GradDiff
    - task_name=tofu_Llama-3.2-3B-Instruct_forget10_GradDiff
    - model=Llama-3.2-3B-Instruct
    - forget_split=forget10
    - retain_split=retain90
    - model.model_args.pretrained_model_name_or_path=open-unlearning/tofu_Llama-3.2-3B-Instruct_full
    - retain_logs_path=saves/eval/tofu_Llama-3.2-3B-Instruct_retain90/TOFU_EVAL.json
    - trainer.args.per_device_train_batch_size=4
    - trainer.args.gradient_accumulation_steps=4
    - trainer.args.ddp_find_unused_parameters=true
    - trainer.args.gradient_checkpointing=true
  job:
    name: train
    chdir: null
    override_dirname: experiment=unlearn/tofu/default.yaml,forget_split=forget10,model.model_args.pretrained_model_name_or_path=open-unlearning/tofu_Llama-3.2-3B-Instruct_full,model=Llama-3.2-3B-Instruct,retain_logs_path=saves/eval/tofu_Llama-3.2-3B-Instruct_retain90/TOFU_EVAL.json,retain_split=retain90,task_name=tofu_Llama-3.2-3B-Instruct_forget10_GradDiff,trainer.args.ddp_find_unused_parameters=true,trainer.args.gradient_accumulation_steps=4,trainer.args.gradient_checkpointing=true,trainer.args.per_device_train_batch_size=4,trainer=GradDiff
    id: ???
    num: ???
    config_name: unlearn.yaml
    env_set: {}
    env_copy: []
    config:
      override_dirname:
        kv_sep: '='
        item_sep: ','
        exclude_keys: []
  runtime:
    version: 1.3.0
    version_base: '1.3'
    cwd: /mnt/nas/slurm_account/thejb/workspace/open-unlearning
    config_sources:
    - path: hydra.conf
      schema: pkg
      provider: hydra
    - path: /mnt/nas/slurm_account/thejb/workspace/open-unlearning/configs
      schema: file
      provider: main
    - path: hydra_plugins.hydra_colorlog.conf
      schema: pkg
      provider: hydra-colorlog
    - path: ''
      schema: structured
      provider: schema
    output_dir: /mnt/nas/slurm_account/thejb/workspace/open-unlearning/saves/unlearn/tofu_Llama-3.2-3B-Instruct_forget10_GradDiff
    choices:
      experiment: unlearn/tofu/default.yaml
      paths: default
      hydra: default
      eval: tofu
      eval/tofu_metrics/../../collator@eval.tofu.metrics.extraction_strength.collators: DataCollatorForSupervisedDatasetwithIndex
      eval/tofu_metrics/../../data/datasets@eval.tofu.metrics.extraction_strength.datasets: TOFU_QA_forget
      eval/tofu_metrics/.@eval.tofu.metrics.privleak.pre_compute.mia_min_k: mia_min_k
      eval/tofu_metrics/./../../collator@eval.tofu.metrics.privleak.pre_compute.mia_min_k.collators: DataCollatorForSupervisedDatasetwithIndex
      eval/tofu_metrics/./../../data/datasets@eval.tofu.metrics.privleak.pre_compute.mia_min_k.datasets: TOFU_MIA
      eval/tofu_metrics/.@eval.tofu.metrics.model_utility.pre_compute.wf_Truth_Ratio: wf_Truth_Ratio
      eval/tofu_metrics/./.@eval.tofu.metrics.model_utility.pre_compute.wf_Truth_Ratio.pre_compute.wf_Q_A_PERT_Prob: wf_Q_A_PERT_Prob
      ? eval/tofu_metrics/././../../collator@eval.tofu.metrics.model_utility.pre_compute.wf_Truth_Ratio.pre_compute.wf_Q_A_PERT_Prob.collators
      : DataCollatorForSupervisedDatasetwithIndex
      ? eval/tofu_metrics/././../../data/datasets@eval.tofu.metrics.model_utility.pre_compute.wf_Truth_Ratio.pre_compute.wf_Q_A_PERT_Prob.datasets
      : TOFU_QA_wf_pert
      eval/tofu_metrics/./.@eval.tofu.metrics.model_utility.pre_compute.wf_Truth_Ratio.pre_compute.wf_Q_A_Prob: wf_Q_A_Prob
      ? eval/tofu_metrics/././../../collator@eval.tofu.metrics.model_utility.pre_compute.wf_Truth_Ratio.pre_compute.wf_Q_A_Prob.collators
      : DataCollatorForSupervisedDatasetwithIndex
      ? eval/tofu_metrics/././../../data/datasets@eval.tofu.metrics.model_utility.pre_compute.wf_Truth_Ratio.pre_compute.wf_Q_A_Prob.datasets
      : TOFU_QA_wf
      eval/tofu_metrics/.@eval.tofu.metrics.model_utility.pre_compute.wf_Q_A_ROUGE: wf_Q_A_ROUGE
      eval/tofu_metrics/./../../generation@eval.tofu.metrics.model_utility.pre_compute.wf_Q_A_ROUGE.generation_args: default
      eval/tofu_metrics/./../../collator@eval.tofu.metrics.model_utility.pre_compute.wf_Q_A_ROUGE.collators: DataCollatorForSupervisedDatasetwithIndex
      eval/tofu_metrics/./../../data/datasets@eval.tofu.metrics.model_utility.pre_compute.wf_Q_A_ROUGE.datasets: TOFU_QA_wf
      eval/tofu_metrics/.@eval.tofu.metrics.model_utility.pre_compute.wf_Q_A_Prob_normalised: wf_Q_A_Prob_normalised
      eval/tofu_metrics/./.@eval.tofu.metrics.model_utility.pre_compute.wf_Q_A_Prob_normalised.pre_compute.wf_Q_A_PERT_Prob: wf_Q_A_PERT_Prob
      ? eval/tofu_metrics/././../../collator@eval.tofu.metrics.model_utility.pre_compute.wf_Q_A_Prob_normalised.pre_compute.wf_Q_A_PERT_Prob.collators
      : DataCollatorForSupervisedDatasetwithIndex
      ? eval/tofu_metrics/././../../data/datasets@eval.tofu.metrics.model_utility.pre_compute.wf_Q_A_Prob_normalised.pre_compute.wf_Q_A_PERT_Prob.datasets
      : TOFU_QA_wf_pert
      eval/tofu_metrics/./.@eval.tofu.metrics.model_utility.pre_compute.wf_Q_A_Prob_normalised.pre_compute.wf_Q_A_Prob: wf_Q_A_Prob
      ? eval/tofu_metrics/././../../collator@eval.tofu.metrics.model_utility.pre_compute.wf_Q_A_Prob_normalised.pre_compute.wf_Q_A_Prob.collators
      : DataCollatorForSupervisedDatasetwithIndex
      ? eval/tofu_metrics/././../../data/datasets@eval.tofu.metrics.model_utility.pre_compute.wf_Q_A_Prob_normalised.pre_compute.wf_Q_A_Prob.datasets
      : TOFU_QA_wf
      eval/tofu_metrics/.@eval.tofu.metrics.model_utility.pre_compute.ra_Truth_Ratio: ra_Truth_Ratio
      eval/tofu_metrics/./.@eval.tofu.metrics.model_utility.pre_compute.ra_Truth_Ratio.pre_compute.ra_Q_A_PERT_Prob: ra_Q_A_PERT_Prob
      ? eval/tofu_metrics/././../../collator@eval.tofu.metrics.model_utility.pre_compute.ra_Truth_Ratio.pre_compute.ra_Q_A_PERT_Prob.collators
      : DataCollatorForSupervisedDatasetwithIndex
      ? eval/tofu_metrics/././../../data/datasets@eval.tofu.metrics.model_utility.pre_compute.ra_Truth_Ratio.pre_compute.ra_Q_A_PERT_Prob.datasets
      : TOFU_QA_ra_pert
      eval/tofu_metrics/./.@eval.tofu.metrics.model_utility.pre_compute.ra_Truth_Ratio.pre_compute.ra_Q_A_Prob: ra_Q_A_Prob
      ? eval/tofu_metrics/././../../collator@eval.tofu.metrics.model_utility.pre_compute.ra_Truth_Ratio.pre_compute.ra_Q_A_Prob.collators
      : DataCollatorForSupervisedDatasetwithIndex
      ? eval/tofu_metrics/././../../data/datasets@eval.tofu.metrics.model_utility.pre_compute.ra_Truth_Ratio.pre_compute.ra_Q_A_Prob.datasets
      : TOFU_QA_ra
      eval/tofu_metrics/.@eval.tofu.metrics.model_utility.pre_compute.ra_Q_A_ROUGE: ra_Q_A_ROUGE
      eval/tofu_metrics/./../../generation@eval.tofu.metrics.model_utility.pre_compute.ra_Q_A_ROUGE.generation_args: default
      eval/tofu_metrics/./../../collator@eval.tofu.metrics.model_utility.pre_compute.ra_Q_A_ROUGE.collators: DataCollatorForSupervisedDatasetwithIndex
      eval/tofu_metrics/./../../data/datasets@eval.tofu.metrics.model_utility.pre_compute.ra_Q_A_ROUGE.datasets: TOFU_QA_ra
      eval/tofu_metrics/.@eval.tofu.metrics.model_utility.pre_compute.ra_Q_A_Prob_normalised: ra_Q_A_Prob_normalised
      eval/tofu_metrics/./.@eval.tofu.metrics.model_utility.pre_compute.ra_Q_A_Prob_normalised.pre_compute.ra_Q_A_PERT_Prob: ra_Q_A_PERT_Prob
      ? eval/tofu_metrics/././../../collator@eval.tofu.metrics.model_utility.pre_compute.ra_Q_A_Prob_normalised.pre_compute.ra_Q_A_PERT_Prob.collators
      : DataCollatorForSupervisedDatasetwithIndex
      ? eval/tofu_metrics/././../../data/datasets@eval.tofu.metrics.model_utility.pre_compute.ra_Q_A_Prob_normalised.pre_compute.ra_Q_A_PERT_Prob.datasets
      : TOFU_QA_ra_pert
      eval/tofu_metrics/./.@eval.tofu.metrics.model_utility.pre_compute.ra_Q_A_Prob_normalised.pre_compute.ra_Q_A_Prob: ra_Q_A_Prob
      ? eval/tofu_metrics/././../../collator@eval.tofu.metrics.model_utility.pre_compute.ra_Q_A_Prob_normalised.pre_compute.ra_Q_A_Prob.collators
      : DataCollatorForSupervisedDatasetwithIndex
      ? eval/tofu_metrics/././../../data/datasets@eval.tofu.metrics.model_utility.pre_compute.ra_Q_A_Prob_normalised.pre_compute.ra_Q_A_Prob.datasets
      : TOFU_QA_ra
      eval/tofu_metrics/.@eval.tofu.metrics.model_utility.pre_compute.retain_Truth_Ratio: retain_Truth_Ratio
      eval/tofu_metrics/./.@eval.tofu.metrics.model_utility.pre_compute.retain_Truth_Ratio.pre_compute.retain_Q_A_PERT_Prob: retain_Q_A_PERT_Prob
      ? eval/tofu_metrics/././../../collator@eval.tofu.metrics.model_utility.pre_compute.retain_Truth_Ratio.pre_compute.retain_Q_A_PERT_Prob.collators
      : DataCollatorForSupervisedDatasetwithIndex
      ? eval/tofu_metrics/././../../data/datasets@eval.tofu.metrics.model_utility.pre_compute.retain_Truth_Ratio.pre_compute.retain_Q_A_PERT_Prob.datasets
      : TOFU_QA_retain_pert
      eval/tofu_metrics/./.@eval.tofu.metrics.model_utility.pre_compute.retain_Truth_Ratio.pre_compute.retain_Q_A_PARA_Prob: retain_Q_A_PARA_Prob
      ? eval/tofu_metrics/././../../collator@eval.tofu.metrics.model_utility.pre_compute.retain_Truth_Ratio.pre_compute.retain_Q_A_PARA_Prob.collators
      : DataCollatorForSupervisedDatasetwithIndex
      ? eval/tofu_metrics/././../../data/datasets@eval.tofu.metrics.model_utility.pre_compute.retain_Truth_Ratio.pre_compute.retain_Q_A_PARA_Prob.datasets
      : TOFU_QA_retain_para
      eval/tofu_metrics/.@eval.tofu.metrics.model_utility.pre_compute.retain_Q_A_ROUGE: retain_Q_A_ROUGE
      eval/tofu_metrics/./../../generation@eval.tofu.metrics.model_utility.pre_compute.retain_Q_A_ROUGE.generation_args: default
      eval/tofu_metrics/./../../collator@eval.tofu.metrics.model_utility.pre_compute.retain_Q_A_ROUGE.collators: DataCollatorForSupervisedDatasetwithIndex
      eval/tofu_metrics/./../../data/datasets@eval.tofu.metrics.model_utility.pre_compute.retain_Q_A_ROUGE.datasets: TOFU_QA_retain_eval
      eval/tofu_metrics/.@eval.tofu.metrics.model_utility.pre_compute.retain_Q_A_Prob: retain_Q_A_Prob
      eval/tofu_metrics/./../../collator@eval.tofu.metrics.model_utility.pre_compute.retain_Q_A_Prob.collators: DataCollatorForSupervisedDatasetwithIndex
      eval/tofu_metrics/./../../data/datasets@eval.tofu.metrics.model_utility.pre_compute.retain_Q_A_Prob.datasets: TOFU_QA_retain_eval
      eval/tofu_metrics/../../generation@eval.tofu.metrics.forget_Q_A_ROUGE.generation_args: default
      eval/tofu_metrics/../../collator@eval.tofu.metrics.forget_Q_A_ROUGE.collators: DataCollatorForSupervisedDatasetwithIndex
      eval/tofu_metrics/../../data/datasets@eval.tofu.metrics.forget_Q_A_ROUGE.datasets: TOFU_QA_forget
      eval/tofu_metrics/../../collator@eval.tofu.metrics.forget_Q_A_Prob.collators: DataCollatorForSupervisedDatasetwithIndex
      eval/tofu_metrics/../../data/datasets@eval.tofu.metrics.forget_Q_A_Prob.datasets: TOFU_QA_forget
      eval/tofu_metrics/.@eval.tofu.metrics.forget_quality.pre_compute.forget_truth_ratio: forget_Truth_Ratio
      eval/tofu_metrics/./.@eval.tofu.metrics.forget_quality.pre_compute.forget_truth_ratio.pre_compute.forget_Q_A_PERT_Prob: forget_Q_A_PERT_Prob
      ? eval/tofu_metrics/././../../collator@eval.tofu.metrics.forget_quality.pre_compute.forget_truth_ratio.pre_compute.forget_Q_A_PERT_Prob.collators
      : DataCollatorForSupervisedDatasetwithIndex
      ? eval/tofu_metrics/././../../data/datasets@eval.tofu.metrics.forget_quality.pre_compute.forget_truth_ratio.pre_compute.forget_Q_A_PERT_Prob.datasets
      : TOFU_QA_forget_pert
      eval/tofu_metrics/./.@eval.tofu.metrics.forget_quality.pre_compute.forget_truth_ratio.pre_compute.forget_Q_A_PARA_Prob: forget_Q_A_PARA_Prob
      ? eval/tofu_metrics/././../../collator@eval.tofu.metrics.forget_quality.pre_compute.forget_truth_ratio.pre_compute.forget_Q_A_PARA_Prob.collators
      : DataCollatorForSupervisedDatasetwithIndex
      ? eval/tofu_metrics/././../../data/datasets@eval.tofu.metrics.forget_quality.pre_compute.forget_truth_ratio.pre_compute.forget_Q_A_PARA_Prob.datasets
      : TOFU_QA_forget_para
      collator: DataCollatorForSupervisedDataset
      data: unlearn
      data/datasets@data.eval: null
      data/datasets@data.retain: TOFU_QA_retain
      data/datasets@data.forget: TOFU_QA_forget
      trainer: GradDiff
      model: Llama-3.2-3B-Instruct
      hydra/env: default
      hydra/callbacks: null
      hydra/job_logging: colorlog
      hydra/hydra_logging: colorlog
      hydra/hydra_help: default
      hydra/help: default
      hydra/sweeper: basic
      hydra/launcher: basic
      hydra/output: default
  verbose: false
--- a/.hydra/overrides.yaml
+++ b/.hydra/overrides.yaml
@@ -0,0 +1,12 @@
 - experiment=unlearn/tofu/default.yaml
 - trainer=GradDiff
 - task_name=tofu_Llama-3.2-3B-Instruct_forget10_GradDiff
 - model=Llama-3.2-3B-Instruct
 - forget_split=forget10
 - retain_split=retain90
 - model.model_args.pretrained_model_name_or_path=open-unlearning/tofu_Llama-3.2-3B-Instruct_full
 - retain_logs_path=saves/eval/tofu_Llama-3.2-3B-Instruct_retain90/TOFU_EVAL.json
 - trainer.args.per_device_train_batch_size=4
 - trainer.args.gradient_accumulation_steps=4
 - trainer.args.ddp_find_unused_parameters=true
 - trainer.args.gradient_checkpointing=true
--- a/GradDiff.log
+++ b/GradDiff.log
@@ -0,0 +1,32 @@
 [2025-05-02 20:02:29,927][model][INFO] - Setting pad_token as eos token: <|eot_id|>
 [2025-05-02 20:02:29,956][model][INFO] - Setting pad_token as eos token: <|eot_id|>
 [2025-05-02 20:02:34,079][evaluator][INFO] - Evaluations stored in the experiment directory: ./saves/unlearn/tofu_Llama-3.2-3B-Instruct_forget10_GradDiff
 [2025-05-02 20:02:34,405][trainer][INFO] - GradDiff Trainer loaded, output_dir: ./saves/unlearn/tofu_Llama-3.2-3B-Instruct_forget10_GradDiff
 [2025-05-02 20:02:34,621][evaluator][INFO] - Evaluations stored in the experiment directory: ./saves/unlearn/tofu_Llama-3.2-3B-Instruct_forget10_GradDiff
 [2025-05-02 20:02:34,806][trainer][INFO] - GradDiff Trainer loaded, output_dir: ./saves/unlearn/tofu_Llama-3.2-3B-Instruct_forget10_GradDiff
 [2025-05-02 20:02:40,756][trainer.base][WARNING] - Custom evaluator can be run with this Trainer only when a single accelerator process is running.
 [2025-05-02 20:04:32,445][trainer.base][WARNING] - Custom evaluator can be run with this Trainer only when a single accelerator process is running.
 [2025-05-02 20:06:23,297][trainer.base][WARNING] - Custom evaluator can be run with this Trainer only when a single accelerator process is running.
 [2025-05-02 20:08:14,092][trainer.base][WARNING] - Custom evaluator can be run with this Trainer only when a single accelerator process is running.
 [2025-05-02 20:10:05,043][trainer.base][WARNING] - Custom evaluator can be run with this Trainer only when a single accelerator process is running.
 [2025-05-02 20:11:55,900][trainer.base][WARNING] - Custom evaluator can be run with this Trainer only when a single accelerator process is running.
 [2025-05-02 20:13:46,806][trainer.base][WARNING] - Custom evaluator can be run with this Trainer only when a single accelerator process is running.
 [2025-05-02 20:15:37,581][trainer.base][WARNING] - Custom evaluator can be run with this Trainer only when a single accelerator process is running.
 [2025-05-02 20:17:28,347][trainer.base][WARNING] - Custom evaluator can be run with this Trainer only when a single accelerator process is running.
 [2025-05-02 20:19:19,097][trainer.base][WARNING] - Custom evaluator can be run with this Trainer only when a single accelerator process is running.
 [2025-05-02 20:20:25,628][trainer.base][WARNING] - Custom evaluator can be run with this Trainer only when a single accelerator process is running.
 [2025-05-02 20:21:00,852][trainer.base][WARNING] - Custom evaluator can be run with this Trainer only when a single accelerator process is running.
 [2025-05-09 19:35:21,761][model][INFO] - Setting pad_token as eos token: <|eot_id|>
 [2025-05-09 19:35:21,902][model][INFO] - Setting pad_token as eos token: <|eot_id|>
 [2025-05-09 19:35:26,849][evaluator][INFO] - Evaluations stored in the experiment directory: ./saves/unlearn/tofu_Llama-3.2-3B-Instruct_forget10_GradDiff
 [2025-05-09 19:35:27,005][evaluator][INFO] - Evaluations stored in the experiment directory: ./saves/unlearn/tofu_Llama-3.2-3B-Instruct_forget10_GradDiff
 [2025-05-09 19:35:27,935][trainer][INFO] - GradDiff Trainer loaded, output_dir: ./saves/unlearn/tofu_Llama-3.2-3B-Instruct_forget10_GradDiff
 [2025-05-09 19:35:28,081][trainer][INFO] - GradDiff Trainer loaded, output_dir: ./saves/unlearn/tofu_Llama-3.2-3B-Instruct_forget10_GradDiff
 [2025-05-13 07:04:46,486][model][INFO] - Setting pad_token as eos token: <|eot_id|>
 [2025-05-13 07:04:46,538][model][INFO] - Setting pad_token as eos token: <|eot_id|>
 [2025-05-13 07:04:51,300][evaluator][INFO] - Evaluations stored in the experiment directory: ./saves/unlearn/tofu_Llama-3.2-3B-Instruct_forget10_GradDiff
 [2025-05-13 07:04:51,963][evaluator][INFO] - Evaluations stored in the experiment directory: ./saves/unlearn/tofu_Llama-3.2-3B-Instruct_forget10_GradDiff
 [2025-05-13 07:04:52,124][trainer][INFO] - GradDiff Trainer loaded, output_dir: ./saves/unlearn/tofu_Llama-3.2-3B-Instruct_forget10_GradDiff
 [2025-05-13 07:04:52,349][trainer][INFO] - GradDiff Trainer loaded, output_dir: ./saves/unlearn/tofu_Llama-3.2-3B-Instruct_forget10_GradDiff
 [2025-05-13 07:04:58,550][trainer.base][WARNING] - Custom evaluator can be run with this Trainer only when a single accelerator process is running.
 [2025-05-13 07:36:54,151][trainer.base][WARNING] - Custom evaluator can be run with this Trainer only when a single accelerator process is running.
--- a/config.json
+++ b/config.json
@@ -0,0 +1,39 @@
 {
  "architectures": [
    "LlamaForCausalLM"
  ],
  "attention_bias": false,
  "attention_dropout": 0.0,
  "bos_token_id": 128000,
  "eos_token_id": [
    128001,
    128008,
    128009
  ],
  "head_dim": 128,
  "hidden_act": "silu",
  "hidden_size": 3072,
  "initializer_range": 0.02,
  "intermediate_size": 8192,
  "max_position_embeddings": 131072,
  "mlp_bias": false,
  "model_type": "llama",
  "num_attention_heads": 24,
  "num_hidden_layers": 28,
  "num_key_value_heads": 8,
  "pretraining_tp": 1,
  "rms_norm_eps": 1e-05,
  "rope_scaling": {
    "factor": 32.0,
    "high_freq_factor": 4.0,
    "low_freq_factor": 1.0,
    "original_max_position_embeddings": 8192,
    "rope_type": "llama3"
  },
  "rope_theta": 500000.0,
  "tie_word_embeddings": true,
  "torch_dtype": "bfloat16",
  "transformers_version": "4.51.3",
  "use_cache": true,
  "vocab_size": 128256
 }
--- a/evals/.hydra/config.yaml
+++ b/evals/.hydra/config.yaml
@@ -0,0 +1,550 @@
 model:
  model_args:
    device_map: cuda
    pretrained_model_name_or_path: saves/unlearn/tofu_Llama-3.2-3B-Instruct_forget10_GradDiff
    attn_implementation: flash_attention_2
    torch_dtype: bfloat16
  tokenizer_args:
    pretrained_model_name_or_path: meta-llama/Llama-3.2-3B-Instruct
  template_args:
    apply_chat_template: true
    system_prompt: You are a helpful assistant.
    system_prompt_with_special_tokens: '<|begin_of_text|><|start_header_id|>system<|end_header_id|>
      You are a helpful assistant.<|eot_id|>'
    user_start_tag: '<|start_header_id|>user<|end_header_id|>
      '
    user_end_tag: <|eot_id|>
    asst_start_tag: '<|start_header_id|>assistant<|end_header_id|>
      '
    asst_end_tag: <|eot_id|>
    date_string: 10 Apr 2025
 mode: eval
 task_name: tofu_Llama-3.2-3B-Instruct_forget10_GradDiff
 seed: 0
 eval:
  tofu:
    metrics:
      forget_quality:
        pre_compute:
          forget_truth_ratio:
            pre_compute:
              forget_Q_A_PARA_Prob:
                datasets:
                  TOFU_QA_forget_para:
                    handler: QADataset
                    args:
                      hf_args:
                        name: ${eval.tofu.forget_split}_perturbed
                        split: train
                        path: locuslab/TOFU
                      question_key: question
                      answer_key: paraphrased_answer
                      max_length: 512
                collators:
                  DataCollatorForSupervisedDataset:
                    handler: DataCollatorForSupervisedDataset
                    args:
                      padding_side: right
                      index: index
                handler: probability
                batch_size: 32
                access_key: correct
              forget_Q_A_PERT_Prob:
                datasets:
                  TOFU_QA_forget_pert:
                    handler: QADataset
                    args:
                      hf_args:
                        name: ${eval.tofu.forget_split}_perturbed
                        split: train
                        path: locuslab/TOFU
                      question_key: question
                      answer_key: perturbed_answer
                      max_length: 512
                collators:
                  DataCollatorForSupervisedDataset:
                    handler: DataCollatorForSupervisedDataset
                    args:
                      padding_side: right
                      index: index
                handler: probability
                batch_size: 32
                access_key: wrong
            handler: truth_ratio
            aggregator: closer_to_1_better
            access_key: forget
        reference_logs:
          retain_model_logs:
            path: ${eval.tofu.retain_logs_path}
            include:
              forget_truth_ratio:
                access_key: retain
        handler: ks_test
      forget_Q_A_Prob:
        datasets:
          TOFU_QA_forget:
            handler: QADataset
            args:
              hf_args:
                name: ${eval.tofu.forget_split}
                split: train
                path: locuslab/TOFU
              question_key: question
              answer_key: answer
              max_length: 512
        collators:
          DataCollatorForSupervisedDataset:
            handler: DataCollatorForSupervisedDataset
            args:
              padding_side: right
              index: index
        handler: probability
        batch_size: 32
      forget_Q_A_ROUGE:
        datasets:
          TOFU_QA_forget:
            handler: QADataset
            args:
              hf_args:
                name: ${eval.tofu.forget_split}
                split: train
                path: locuslab/TOFU
              question_key: question
              answer_key: answer
              max_length: 512
              predict_with_generate: true
        collators:
          DataCollatorForSupervisedDataset:
            handler: DataCollatorForSupervisedDataset
            args:
              padding_side: left
              index: index
        generation_args:
          do_sample: false
          top_p: null
          temperature: null
          max_new_tokens: 200
          use_cache: true
        handler: rouge
        rouge_type: rougeL_recall
        batch_size: 32
      model_utility:
        pre_compute:
          retain_Q_A_Prob:
            datasets:
              TOFU_QA_retain_eval:
                handler: QADataset
                args:
                  hf_args:
                    name: retain_perturbed
                    split: train
                    path: locuslab/TOFU
                  question_key: question
                  answer_key: answer
                  max_length: 512
            collators:
              DataCollatorForSupervisedDataset:
                handler: DataCollatorForSupervisedDataset
                args:
                  padding_side: right
                  index: index
            handler: probability
            batch_size: 32
          retain_Q_A_ROUGE:
            datasets:
              TOFU_QA_retain_eval:
                handler: QADataset
                args:
                  hf_args:
                    name: retain_perturbed
                    split: train
                    path: locuslab/TOFU
                  question_key: question
                  answer_key: answer
                  max_length: 512
                  predict_with_generate: true
            collators:
              DataCollatorForSupervisedDataset:
                handler: DataCollatorForSupervisedDataset
                args:
                  padding_side: left
                  index: index
            generation_args:
              do_sample: false
              top_p: null
              temperature: null
              max_new_tokens: 200
              use_cache: true
            handler: rouge
            rouge_type: rougeL_recall
            batch_size: 32
          retain_Truth_Ratio:
            pre_compute:
              retain_Q_A_PARA_Prob:
                datasets:
                  TOFU_QA_retain_para:
                    handler: QADataset
                    args:
                      hf_args:
                        name: retain_perturbed
                        split: train
                        path: locuslab/TOFU
                      question_key: question
                      answer_key: paraphrased_answer
                      max_length: 512
                collators:
                  DataCollatorForSupervisedDataset:
                    handler: DataCollatorForSupervisedDataset
                    args:
                      padding_side: right
                      index: index
                handler: probability
                batch_size: 32
                access_key: correct
              retain_Q_A_PERT_Prob:
                datasets:
                  TOFU_QA_retain_pert:
                    handler: QADataset
                    args:
                      hf_args:
                        name: retain_perturbed
                        split: train
                        path: locuslab/TOFU
                      question_key: question
                      answer_key: perturbed_answer
                      max_length: 512
                collators:
                  DataCollatorForSupervisedDataset:
                    handler: DataCollatorForSupervisedDataset
                    args:
                      padding_side: right
                      index: index
                handler: probability
                batch_size: 32
                access_key: wrong
            handler: truth_ratio
            aggregator: true_better
          ra_Q_A_Prob_normalised:
            pre_compute:
              ra_Q_A_Prob:
                datasets:
                  TOFU_QA_ra:
                    handler: QADataset
                    args:
                      hf_args:
                        name: real_authors_perturbed
                        split: train
                        path: locuslab/TOFU
                      question_key: question
                      answer_key: answer
                      max_length: 512
                collators:
                  DataCollatorForSupervisedDataset:
                    handler: DataCollatorForSupervisedDataset
                    args:
                      padding_side: right
                      index: index
                handler: probability
                batch_size: 32
                access_key: correct
              ra_Q_A_PERT_Prob:
                datasets:
                  TOFU_QA_ra_pert:
                    handler: QADataset
                    args:
                      hf_args:
                        name: real_authors_perturbed
                        split: train
                        path: locuslab/TOFU
                      question_key: question
                      answer_key: perturbed_answer
                      max_length: 512
                collators:
                  DataCollatorForSupervisedDataset:
                    handler: DataCollatorForSupervisedDataset
                    args:
                      padding_side: right
                      index: index
                handler: probability
                batch_size: 32
                access_key: wrong
            handler: probability_w_options
          ra_Q_A_ROUGE:
            datasets:
              TOFU_QA_ra:
                handler: QADataset
                args:
                  hf_args:
                    name: real_authors_perturbed
                    split: train
                    path: locuslab/TOFU
                  question_key: question
                  answer_key: answer
                  max_length: 512
                  predict_with_generate: true
            collators:
              DataCollatorForSupervisedDataset:
                handler: DataCollatorForSupervisedDataset
                args:
                  padding_side: left
                  index: index
            generation_args:
              do_sample: false
              top_p: null
              temperature: null
              max_new_tokens: 200
              use_cache: true
            handler: rouge
            rouge_type: rougeL_recall
            batch_size: 32
          ra_Truth_Ratio:
            pre_compute:
              ra_Q_A_Prob:
                datasets:
                  TOFU_QA_ra:
                    handler: QADataset
                    args:
                      hf_args:
                        name: real_authors_perturbed
                        split: train
                        path: locuslab/TOFU
                      question_key: question
                      answer_key: answer
                      max_length: 512
                collators:
                  DataCollatorForSupervisedDataset:
                    handler: DataCollatorForSupervisedDataset
                    args:
                      padding_side: right
                      index: index
                handler: probability
                batch_size: 32
                access_key: correct
              ra_Q_A_PERT_Prob:
                datasets:
                  TOFU_QA_ra_pert:
                    handler: QADataset
                    args:
                      hf_args:
                        name: real_authors_perturbed
                        split: train
                        path: locuslab/TOFU
                      question_key: question
                      answer_key: perturbed_answer
                      max_length: 512
                collators:
                  DataCollatorForSupervisedDataset:
                    handler: DataCollatorForSupervisedDataset
                    args:
                      padding_side: right
                      index: index
                handler: probability
                batch_size: 32
                access_key: wrong
            handler: truth_ratio
            aggregator: true_better
          wf_Q_A_Prob_normalised:
            pre_compute:
              wf_Q_A_Prob:
                datasets:
                  TOFU_QA_wf:
                    handler: QADataset
                    args:
                      hf_args:
                        name: world_facts_perturbed
                        split: train
                        path: locuslab/TOFU
                      question_key: question
                      answer_key: answer
                      max_length: 512
                collators:
                  DataCollatorForSupervisedDataset:
                    handler: DataCollatorForSupervisedDataset
                    args:
                      padding_side: right
                      index: index
                handler: probability
                batch_size: 32
                access_key: correct
              wf_Q_A_PERT_Prob:
                datasets:
                  TOFU_QA_wf_pert:
                    handler: QADataset
                    args:
                      hf_args:
                        name: world_facts_perturbed
                        split: train
                        path: locuslab/TOFU
                      question_key: question
                      answer_key: perturbed_answer
                      max_length: 512
                collators:
                  DataCollatorForSupervisedDataset:
                    handler: DataCollatorForSupervisedDataset
                    args:
                      padding_side: right
                      index: index
                handler: probability
                batch_size: 32
                access_key: wrong
            handler: probability_w_options
          wf_Q_A_ROUGE:
            datasets:
              TOFU_QA_wf:
                handler: QADataset
                args:
                  hf_args:
                    name: world_facts_perturbed
                    split: train
                    path: locuslab/TOFU
                  question_key: question
                  answer_key: answer
                  max_length: 512
                  predict_with_generate: true
            collators:
              DataCollatorForSupervisedDataset:
                handler: DataCollatorForSupervisedDataset
                args:
                  padding_side: left
                  index: index
            generation_args:
              do_sample: false
              top_p: null
              temperature: null
              max_new_tokens: 200
              use_cache: true
            handler: rouge
            rouge_type: rougeL_recall
            batch_size: 32
          wf_Truth_Ratio:
            pre_compute:
              wf_Q_A_Prob:
                datasets:
                  TOFU_QA_wf:
                    handler: QADataset
                    args:
                      hf_args:
                        name: world_facts_perturbed
                        split: train
                        path: locuslab/TOFU
                      question_key: question
                      answer_key: answer
                      max_length: 512
                collators:
                  DataCollatorForSupervisedDataset:
                    handler: DataCollatorForSupervisedDataset
                    args:
                      padding_side: right
                      index: index
                handler: probability
                batch_size: 32
                access_key: correct
              wf_Q_A_PERT_Prob:
                datasets:
                  TOFU_QA_wf_pert:
                    handler: QADataset
                    args:
                      hf_args:
                        name: world_facts_perturbed
                        split: train
                        path: locuslab/TOFU
                      question_key: question
                      answer_key: perturbed_answer
                      max_length: 512
                collators:
                  DataCollatorForSupervisedDataset:
                    handler: DataCollatorForSupervisedDataset
                    args:
                      padding_side: right
                      index: index
                handler: probability
                batch_size: 32
                access_key: wrong
            handler: truth_ratio
            aggregator: true_better
        handler: hm_aggregate
      privleak:
        pre_compute:
          mia_min_k:
            datasets:
              TOFU_QA_forget:
                access_key: forget
                handler: QADataset
                args:
                  hf_args:
                    name: ${eval.tofu.forget_split}
                    split: train
                    path: locuslab/TOFU
                  question_key: question
                  answer_key: answer
                  max_length: 512
              TOFU_QA_holdout:
                access_key: holdout
                handler: QADataset
                args:
                  hf_args:
                    name: ${eval.tofu.holdout_split}
                    path: locuslab/TOFU
                    split: train
                  question_key: question
                  answer_key: answer
                  max_length: 512
            collators:
              DataCollatorForSupervisedDataset:
                handler: DataCollatorForSupervisedDataset
                args:
                  padding_side: right
                  index: index
            batch_size: 32
            handler: mia_min_k
            k: 0.4
            access_key: forget
        reference_logs:
          retain_model_logs:
            path: ${eval.tofu.retain_logs_path}
            include:
              mia_min_k:
                access_key: retain
        handler: privleak
        ref_value: 0.5
      extraction_strength:
        datasets:
          TOFU_QA_forget:
            handler: QADataset
            args:
              hf_args:
                name: ${eval.tofu.forget_split}
                split: train
                path: locuslab/TOFU
              question_key: question
              answer_key: answer
              max_length: 512
        collators:
          DataCollatorForSupervisedDataset:
            handler: DataCollatorForSupervisedDataset
            args:
              padding_side: right
              index: index
        handler: extraction_strength
        batch_size: 32
    handler: TOFUEvaluator
    output_dir: ${paths.output_dir}
    overwrite: false
    forget_split: ${forget_split}
    holdout_split: ${holdout_split}
    retain_logs_path: ${retain_logs_path}
 paths:
  root_dir: .
  data_dir: ${paths.root_dir}/data/
  datasets: ${paths.root_dir}/configs/data/datasets
  output_dir: saves/unlearn/tofu_Llama-3.2-3B-Instruct_forget10_GradDiff/evals
  work_dir: ${hydra:runtime.cwd}
 forget_split: forget10
 holdout_split: holdout10
 retain_logs_path: saves/eval/tofu_Llama-3.2-3B-Instruct_retain90/TOFU_EVAL.json
--- a/evals/.hydra/hydra.yaml
+++ b/evals/.hydra/hydra.yaml
@@ -0,0 +1,269 @@
 hydra:
  run:
    dir: ${paths.output_dir}
  sweep:
    dir: multirun/${now:%Y-%m-%d}/${now:%H-%M-%S}
    subdir: ${hydra.job.num}
  launcher:
    _target_: hydra._internal.core_plugins.basic_launcher.BasicLauncher
  sweeper:
    _target_: hydra._internal.core_plugins.basic_sweeper.BasicSweeper
    max_batch_size: null
    params: null
  help:
    app_name: ${hydra.job.name}
    header: '${hydra.help.app_name} is powered by Hydra.
      '
    footer: 'Powered by Hydra (https://hydra.cc)
      Use --hydra-help to view Hydra specific help
      '
    template: '${hydra.help.header}
      == Configuration groups ==
      Compose your configuration from those groups (group=option)
      $APP_CONFIG_GROUPS
      == Config ==
      Override anything in the config (foo.bar=value)
      $CONFIG
      ${hydra.help.footer}
      '
  hydra_help:
    template: 'Hydra (${hydra.runtime.version})
      See https://hydra.cc for more info.
      == Flags ==
      $FLAGS_HELP
      == Configuration groups ==
      Compose your configuration from those groups (For example, append hydra/job_logging=disabled
      to command line)
      $HYDRA_CONFIG_GROUPS
      Use ''--cfg hydra'' to Show the Hydra config.
      '
    hydra_help: ???
  hydra_logging:
    version: 1
    formatters:
      colorlog:
        (): colorlog.ColoredFormatter
        format: '[%(cyan)s%(asctime)s%(reset)s][%(purple)sHYDRA%(reset)s] %(message)s'
    handlers:
      console:
        class: logging.StreamHandler
        formatter: colorlog
        stream: ext://sys.stdout
    root:
      level: INFO
      handlers:
      - console
    disable_existing_loggers: false
  job_logging:
    version: 1
    formatters:
      simple:
        format: '[%(asctime)s][%(name)s][%(levelname)s] - %(message)s'
      colorlog:
        (): colorlog.ColoredFormatter
        format: '[%(cyan)s%(asctime)s%(reset)s][%(blue)s%(name)s%(reset)s][%(log_color)s%(levelname)s%(reset)s]
          - %(message)s'
        log_colors:
          DEBUG: purple
          INFO: green
          WARNING: yellow
          ERROR: red
          CRITICAL: red
    handlers:
      console:
        class: logging.StreamHandler
        formatter: colorlog
        stream: ext://sys.stdout
      file:
        class: logging.FileHandler
        formatter: simple
        filename: ${hydra.runtime.output_dir}/eval.log
    root:
      level: INFO
      handlers:
      - console
      - file
    disable_existing_loggers: false
  env: {}
  mode: RUN
  searchpath: []
  callbacks: {}
  output_subdir: .hydra
  overrides:
    hydra:
    - hydra.mode=RUN
    task:
    - experiment=eval/tofu/default.yaml
    - forget_split=forget10
    - holdout_split=holdout10
    - model=Llama-3.2-3B-Instruct
    - task_name=tofu_Llama-3.2-3B-Instruct_forget10_GradDiff
    - model.model_args.pretrained_model_name_or_path=saves/unlearn/tofu_Llama-3.2-3B-Instruct_forget10_GradDiff
    - paths.output_dir=saves/unlearn/tofu_Llama-3.2-3B-Instruct_forget10_GradDiff/evals
    - retain_logs_path=saves/eval/tofu_Llama-3.2-3B-Instruct_retain90/TOFU_EVAL.json
  job:
    name: eval
    chdir: null
    override_dirname: experiment=eval/tofu/default.yaml,forget_split=forget10,holdout_split=holdout10,model.model_args.pretrained_model_name_or_path=saves/unlearn/tofu_Llama-3.2-3B-Instruct_forget10_GradDiff,model=Llama-3.2-3B-Instruct,paths.output_dir=saves/unlearn/tofu_Llama-3.2-3B-Instruct_forget10_GradDiff/evals,retain_logs_path=saves/eval/tofu_Llama-3.2-3B-Instruct_retain90/TOFU_EVAL.json,task_name=tofu_Llama-3.2-3B-Instruct_forget10_GradDiff
    id: ???
    num: ???
    config_name: eval.yaml
    env_set: {}
    env_copy: []
    config:
      override_dirname:
        kv_sep: '='
        item_sep: ','
        exclude_keys: []
  runtime:
    version: 1.3.0
    version_base: '1.3'
    cwd: /mnt/nas/slurm_account/thejb/workspace/open-unlearning
    config_sources:
    - path: hydra.conf
      schema: pkg
      provider: hydra
    - path: /mnt/nas/slurm_account/thejb/workspace/open-unlearning/configs
      schema: file
      provider: main
    - path: hydra_plugins.hydra_colorlog.conf
      schema: pkg
      provider: hydra-colorlog
    - path: ''
      schema: structured
      provider: schema
    output_dir: /mnt/nas/slurm_account/thejb/workspace/open-unlearning/saves/unlearn/tofu_Llama-3.2-3B-Instruct_forget10_GradDiff/evals
    choices:
      experiment: eval/tofu/default.yaml
      hydra: eval
      paths: default
      eval: tofu
      eval/tofu_metrics/../../collator@eval.tofu.metrics.extraction_strength.collators: DataCollatorForSupervisedDatasetwithIndex
      eval/tofu_metrics/../../data/datasets@eval.tofu.metrics.extraction_strength.datasets: TOFU_QA_forget
      eval/tofu_metrics/.@eval.tofu.metrics.privleak.pre_compute.mia_min_k: mia_min_k
      eval/tofu_metrics/./../../collator@eval.tofu.metrics.privleak.pre_compute.mia_min_k.collators: DataCollatorForSupervisedDatasetwithIndex
      eval/tofu_metrics/./../../data/datasets@eval.tofu.metrics.privleak.pre_compute.mia_min_k.datasets: TOFU_MIA
      eval/tofu_metrics/.@eval.tofu.metrics.model_utility.pre_compute.wf_Truth_Ratio: wf_Truth_Ratio
      eval/tofu_metrics/./.@eval.tofu.metrics.model_utility.pre_compute.wf_Truth_Ratio.pre_compute.wf_Q_A_PERT_Prob: wf_Q_A_PERT_Prob
      ? eval/tofu_metrics/././../../collator@eval.tofu.metrics.model_utility.pre_compute.wf_Truth_Ratio.pre_compute.wf_Q_A_PERT_Prob.collators
      : DataCollatorForSupervisedDatasetwithIndex
      ? eval/tofu_metrics/././../../data/datasets@eval.tofu.metrics.model_utility.pre_compute.wf_Truth_Ratio.pre_compute.wf_Q_A_PERT_Prob.datasets
      : TOFU_QA_wf_pert
      eval/tofu_metrics/./.@eval.tofu.metrics.model_utility.pre_compute.wf_Truth_Ratio.pre_compute.wf_Q_A_Prob: wf_Q_A_Prob
      ? eval/tofu_metrics/././../../collator@eval.tofu.metrics.model_utility.pre_compute.wf_Truth_Ratio.pre_compute.wf_Q_A_Prob.collators
      : DataCollatorForSupervisedDatasetwithIndex
      ? eval/tofu_metrics/././../../data/datasets@eval.tofu.metrics.model_utility.pre_compute.wf_Truth_Ratio.pre_compute.wf_Q_A_Prob.datasets
      : TOFU_QA_wf
      eval/tofu_metrics/.@eval.tofu.metrics.model_utility.pre_compute.wf_Q_A_ROUGE: wf_Q_A_ROUGE
      eval/tofu_metrics/./../../generation@eval.tofu.metrics.model_utility.pre_compute.wf_Q_A_ROUGE.generation_args: default
      eval/tofu_metrics/./../../collator@eval.tofu.metrics.model_utility.pre_compute.wf_Q_A_ROUGE.collators: DataCollatorForSupervisedDatasetwithIndex
      eval/tofu_metrics/./../../data/datasets@eval.tofu.metrics.model_utility.pre_compute.wf_Q_A_ROUGE.datasets: TOFU_QA_wf
      eval/tofu_metrics/.@eval.tofu.metrics.model_utility.pre_compute.wf_Q_A_Prob_normalised: wf_Q_A_Prob_normalised
      eval/tofu_metrics/./.@eval.tofu.metrics.model_utility.pre_compute.wf_Q_A_Prob_normalised.pre_compute.wf_Q_A_PERT_Prob: wf_Q_A_PERT_Prob
      ? eval/tofu_metrics/././../../collator@eval.tofu.metrics.model_utility.pre_compute.wf_Q_A_Prob_normalised.pre_compute.wf_Q_A_PERT_Prob.collators
      : DataCollatorForSupervisedDatasetwithIndex
      ? eval/tofu_metrics/././../../data/datasets@eval.tofu.metrics.model_utility.pre_compute.wf_Q_A_Prob_normalised.pre_compute.wf_Q_A_PERT_Prob.datasets
      : TOFU_QA_wf_pert
      eval/tofu_metrics/./.@eval.tofu.metrics.model_utility.pre_compute.wf_Q_A_Prob_normalised.pre_compute.wf_Q_A_Prob: wf_Q_A_Prob
      ? eval/tofu_metrics/././../../collator@eval.tofu.metrics.model_utility.pre_compute.wf_Q_A_Prob_normalised.pre_compute.wf_Q_A_Prob.collators
      : DataCollatorForSupervisedDatasetwithIndex
      ? eval/tofu_metrics/././../../data/datasets@eval.tofu.metrics.model_utility.pre_compute.wf_Q_A_Prob_normalised.pre_compute.wf_Q_A_Prob.datasets
      : TOFU_QA_wf
      eval/tofu_metrics/.@eval.tofu.metrics.model_utility.pre_compute.ra_Truth_Ratio: ra_Truth_Ratio
      eval/tofu_metrics/./.@eval.tofu.metrics.model_utility.pre_compute.ra_Truth_Ratio.pre_compute.ra_Q_A_PERT_Prob: ra_Q_A_PERT_Prob
      ? eval/tofu_metrics/././../../collator@eval.tofu.metrics.model_utility.pre_compute.ra_Truth_Ratio.pre_compute.ra_Q_A_PERT_Prob.collators
      : DataCollatorForSupervisedDatasetwithIndex
      ? eval/tofu_metrics/././../../data/datasets@eval.tofu.metrics.model_utility.pre_compute.ra_Truth_Ratio.pre_compute.ra_Q_A_PERT_Prob.datasets
      : TOFU_QA_ra_pert
      eval/tofu_metrics/./.@eval.tofu.metrics.model_utility.pre_compute.ra_Truth_Ratio.pre_compute.ra_Q_A_Prob: ra_Q_A_Prob
      ? eval/tofu_metrics/././../../collator@eval.tofu.metrics.model_utility.pre_compute.ra_Truth_Ratio.pre_compute.ra_Q_A_Prob.collators
      : DataCollatorForSupervisedDatasetwithIndex
      ? eval/tofu_metrics/././../../data/datasets@eval.tofu.metrics.model_utility.pre_compute.ra_Truth_Ratio.pre_compute.ra_Q_A_Prob.datasets
      : TOFU_QA_ra
      eval/tofu_metrics/.@eval.tofu.metrics.model_utility.pre_compute.ra_Q_A_ROUGE: ra_Q_A_ROUGE
      eval/tofu_metrics/./../../generation@eval.tofu.metrics.model_utility.pre_compute.ra_Q_A_ROUGE.generation_args: default
      eval/tofu_metrics/./../../collator@eval.tofu.metrics.model_utility.pre_compute.ra_Q_A_ROUGE.collators: DataCollatorForSupervisedDatasetwithIndex
      eval/tofu_metrics/./../../data/datasets@eval.tofu.metrics.model_utility.pre_compute.ra_Q_A_ROUGE.datasets: TOFU_QA_ra
      eval/tofu_metrics/.@eval.tofu.metrics.model_utility.pre_compute.ra_Q_A_Prob_normalised: ra_Q_A_Prob_normalised
      eval/tofu_metrics/./.@eval.tofu.metrics.model_utility.pre_compute.ra_Q_A_Prob_normalised.pre_compute.ra_Q_A_PERT_Prob: ra_Q_A_PERT_Prob
      ? eval/tofu_metrics/././../../collator@eval.tofu.metrics.model_utility.pre_compute.ra_Q_A_Prob_normalised.pre_compute.ra_Q_A_PERT_Prob.collators
      : DataCollatorForSupervisedDatasetwithIndex
      ? eval/tofu_metrics/././../../data/datasets@eval.tofu.metrics.model_utility.pre_compute.ra_Q_A_Prob_normalised.pre_compute.ra_Q_A_PERT_Prob.datasets
      : TOFU_QA_ra_pert
      eval/tofu_metrics/./.@eval.tofu.metrics.model_utility.pre_compute.ra_Q_A_Prob_normalised.pre_compute.ra_Q_A_Prob: ra_Q_A_Prob
      ? eval/tofu_metrics/././../../collator@eval.tofu.metrics.model_utility.pre_compute.ra_Q_A_Prob_normalised.pre_compute.ra_Q_A_Prob.collators
      : DataCollatorForSupervisedDatasetwithIndex
      ? eval/tofu_metrics/././../../data/datasets@eval.tofu.metrics.model_utility.pre_compute.ra_Q_A_Prob_normalised.pre_compute.ra_Q_A_Prob.datasets
      : TOFU_QA_ra
      eval/tofu_metrics/.@eval.tofu.metrics.model_utility.pre_compute.retain_Truth_Ratio: retain_Truth_Ratio
      eval/tofu_metrics/./.@eval.tofu.metrics.model_utility.pre_compute.retain_Truth_Ratio.pre_compute.retain_Q_A_PERT_Prob: retain_Q_A_PERT_Prob
      ? eval/tofu_metrics/././../../collator@eval.tofu.metrics.model_utility.pre_compute.retain_Truth_Ratio.pre_compute.retain_Q_A_PERT_Prob.collators
      : DataCollatorForSupervisedDatasetwithIndex
      ? eval/tofu_metrics/././../../data/datasets@eval.tofu.metrics.model_utility.pre_compute.retain_Truth_Ratio.pre_compute.retain_Q_A_PERT_Prob.datasets
      : TOFU_QA_retain_pert
      eval/tofu_metrics/./.@eval.tofu.metrics.model_utility.pre_compute.retain_Truth_Ratio.pre_compute.retain_Q_A_PARA_Prob: retain_Q_A_PARA_Prob
      ? eval/tofu_metrics/././../../collator@eval.tofu.metrics.model_utility.pre_compute.retain_Truth_Ratio.pre_compute.retain_Q_A_PARA_Prob.collators
      : DataCollatorForSupervisedDatasetwithIndex
      ? eval/tofu_metrics/././../../data/datasets@eval.tofu.metrics.model_utility.pre_compute.retain_Truth_Ratio.pre_compute.retain_Q_A_PARA_Prob.datasets
      : TOFU_QA_retain_para
      eval/tofu_metrics/.@eval.tofu.metrics.model_utility.pre_compute.retain_Q_A_ROUGE: retain_Q_A_ROUGE
      eval/tofu_metrics/./../../generation@eval.tofu.metrics.model_utility.pre_compute.retain_Q_A_ROUGE.generation_args: default
      eval/tofu_metrics/./../../collator@eval.tofu.metrics.model_utility.pre_compute.retain_Q_A_ROUGE.collators: DataCollatorForSupervisedDatasetwithIndex
      eval/tofu_metrics/./../../data/datasets@eval.tofu.metrics.model_utility.pre_compute.retain_Q_A_ROUGE.datasets: TOFU_QA_retain_eval
      eval/tofu_metrics/.@eval.tofu.metrics.model_utility.pre_compute.retain_Q_A_Prob: retain_Q_A_Prob
      eval/tofu_metrics/./../../collator@eval.tofu.metrics.model_utility.pre_compute.retain_Q_A_Prob.collators: DataCollatorForSupervisedDatasetwithIndex
      eval/tofu_metrics/./../../data/datasets@eval.tofu.metrics.model_utility.pre_compute.retain_Q_A_Prob.datasets: TOFU_QA_retain_eval
      eval/tofu_metrics/../../generation@eval.tofu.metrics.forget_Q_A_ROUGE.generation_args: default
      eval/tofu_metrics/../../collator@eval.tofu.metrics.forget_Q_A_ROUGE.collators: DataCollatorForSupervisedDatasetwithIndex
      eval/tofu_metrics/../../data/datasets@eval.tofu.metrics.forget_Q_A_ROUGE.datasets: TOFU_QA_forget
      eval/tofu_metrics/../../collator@eval.tofu.metrics.forget_Q_A_Prob.collators: DataCollatorForSupervisedDatasetwithIndex
      eval/tofu_metrics/../../data/datasets@eval.tofu.metrics.forget_Q_A_Prob.datasets: TOFU_QA_forget
      eval/tofu_metrics/.@eval.tofu.metrics.forget_quality.pre_compute.forget_truth_ratio: forget_Truth_Ratio
      eval/tofu_metrics/./.@eval.tofu.metrics.forget_quality.pre_compute.forget_truth_ratio.pre_compute.forget_Q_A_PERT_Prob: forget_Q_A_PERT_Prob
      ? eval/tofu_metrics/././../../collator@eval.tofu.metrics.forget_quality.pre_compute.forget_truth_ratio.pre_compute.forget_Q_A_PERT_Prob.collators
      : DataCollatorForSupervisedDatasetwithIndex
      ? eval/tofu_metrics/././../../data/datasets@eval.tofu.metrics.forget_quality.pre_compute.forget_truth_ratio.pre_compute.forget_Q_A_PERT_Prob.datasets
      : TOFU_QA_forget_pert
      eval/tofu_metrics/./.@eval.tofu.metrics.forget_quality.pre_compute.forget_truth_ratio.pre_compute.forget_Q_A_PARA_Prob: forget_Q_A_PARA_Prob
      ? eval/tofu_metrics/././../../collator@eval.tofu.metrics.forget_quality.pre_compute.forget_truth_ratio.pre_compute.forget_Q_A_PARA_Prob.collators
      : DataCollatorForSupervisedDatasetwithIndex
      ? eval/tofu_metrics/././../../data/datasets@eval.tofu.metrics.forget_quality.pre_compute.forget_truth_ratio.pre_compute.forget_Q_A_PARA_Prob.datasets
      : TOFU_QA_forget_para
      model: Llama-3.2-3B-Instruct
      hydra/env: default
      hydra/callbacks: null
      hydra/job_logging: colorlog
      hydra/hydra_logging: colorlog
      hydra/hydra_help: default
      hydra/help: default
      hydra/sweeper: basic
      hydra/launcher: basic
      hydra/output: default
  verbose: false
--- a/evals/.hydra/overrides.yaml
+++ b/evals/.hydra/overrides.yaml
@@ -0,0 +1,8 @@
 - experiment=eval/tofu/default.yaml
 - forget_split=forget10
 - holdout_split=holdout10
 - model=Llama-3.2-3B-Instruct
 - task_name=tofu_Llama-3.2-3B-Instruct_forget10_GradDiff
 - model.model_args.pretrained_model_name_or_path=saves/unlearn/tofu_Llama-3.2-3B-Instruct_forget10_GradDiff
 - paths.output_dir=saves/unlearn/tofu_Llama-3.2-3B-Instruct_forget10_GradDiff/evals
 - retain_logs_path=saves/eval/tofu_Llama-3.2-3B-Instruct_retain90/TOFU_EVAL.json
--- a/evals/TOFU_EVAL.json
+++ b/evals/TOFU_EVAL.json
--- a/evals/TOFU_SUMMARY.json
+++ b/evals/TOFU_SUMMARY.json
@@ -0,0 +1,8 @@
 {
    "extraction_strength": 0.03250892997513522,
    "forget_Q_A_Prob": 4.883993844752212e-09,
    "forget_Q_A_ROUGE": 0.005082328795915088,
    "forget_quality": 2.50770024871112e-208,
    "model_utility": 0.5869124818129423,
    "privleak": 64.70836985162877
 }
--- a/evals/eval.log
+++ b/evals/eval.log
@@ -0,0 +1,82 @@
 [2025-05-02 20:21:25,819][model][INFO] - Setting pad_token as eos token: <|eot_id|>
 [2025-05-02 20:21:25,822][evaluator][INFO] - Evaluations stored in the experiment directory: saves/unlearn/tofu_Llama-3.2-3B-Instruct_forget10_GradDiff/evals
 [2025-05-02 20:21:25,824][evaluator][INFO] - ***** Running TOFU evaluation suite *****
 [2025-05-02 20:21:25,824][evaluator][INFO] - Fine-grained evaluations will be saved to: saves/unlearn/tofu_Llama-3.2-3B-Instruct_forget10_GradDiff/evals/TOFU_EVAL.json
 [2025-05-02 20:21:25,824][evaluator][INFO] - Aggregated evaluations will be summarised in: saves/unlearn/tofu_Llama-3.2-3B-Instruct_forget10_GradDiff/evals/TOFU_SUMMARY.json
 [2025-05-02 20:21:29,284][metrics][INFO] - Loading evaluations from saves/eval/tofu_Llama-3.2-3B-Instruct_retain90/TOFU_EVAL.json
 [2025-05-02 20:21:29,296][metrics][INFO] - Evaluating forget_Q_A_PARA_Prob
 [2025-05-02 20:21:38,558][metrics][INFO] - Loading evaluations from saves/eval/tofu_Llama-3.2-3B-Instruct_retain90/TOFU_EVAL.json
 [2025-05-02 20:21:38,571][metrics][INFO] - Evaluating forget_Q_A_PERT_Prob
 [2025-05-02 20:22:10,457][metrics][INFO] - Loading evaluations from saves/eval/tofu_Llama-3.2-3B-Instruct_retain90/TOFU_EVAL.json
 [2025-05-02 20:22:10,469][metrics][INFO] - Evaluating forget_truth_ratio
 [2025-05-02 20:22:10,470][metrics][INFO] - Loading evaluations from saves/eval/tofu_Llama-3.2-3B-Instruct_retain90/TOFU_EVAL.json
 [2025-05-02 20:22:10,479][metrics][INFO] - Evaluating forget_quality
 [2025-05-02 20:22:10,481][evaluator][INFO] - Result for metric forget_quality:	2.50770024871112e-208
 [2025-05-02 20:22:12,629][metrics][INFO] - Evaluating forget_Q_A_Prob
 [2025-05-02 20:22:19,170][evaluator][INFO] - Result for metric forget_Q_A_Prob:	4.883993844752212e-09
 [2025-05-02 20:22:20,912][metrics][INFO] - Evaluating forget_Q_A_ROUGE
 [2025-05-02 20:23:30,110][evaluator][INFO] - Result for metric forget_Q_A_ROUGE:	0.005082328795915088
 [2025-05-02 20:23:32,243][metrics][INFO] - Evaluating retain_Q_A_Prob
 [2025-05-02 20:23:39,680][metrics][INFO] - Evaluating retain_Q_A_ROUGE
 [2025-05-02 20:24:24,676][metrics][INFO] - Evaluating retain_Q_A_PARA_Prob
 [2025-05-02 20:24:32,393][metrics][INFO] - Evaluating retain_Q_A_PERT_Prob
 [2025-05-02 20:25:01,809][metrics][INFO] - Evaluating retain_Truth_Ratio
 [2025-05-02 20:25:03,699][metrics][INFO] - Evaluating ra_Q_A_Prob
 [2025-05-02 20:25:06,491][metrics][INFO] - Evaluating ra_Q_A_PERT_Prob
 [2025-05-02 20:25:09,527][metrics][INFO] - Evaluating ra_Q_A_Prob_normalised
 [2025-05-02 20:25:10,802][metrics][INFO] - Evaluating ra_Q_A_ROUGE
 [2025-05-02 20:25:31,069][metrics][INFO] - Skipping ra_Truth_Ratio's precompute ra_Q_A_Prob, already evaluated.
 [2025-05-02 20:25:31,069][metrics][INFO] - Skipping ra_Truth_Ratio's precompute ra_Q_A_PERT_Prob, already evaluated.
 [2025-05-02 20:25:31,069][metrics][INFO] - Evaluating ra_Truth_Ratio
 [2025-05-02 20:25:32,784][metrics][INFO] - Evaluating wf_Q_A_Prob
 [2025-05-02 20:25:35,081][metrics][INFO] - Evaluating wf_Q_A_PERT_Prob
 [2025-05-02 20:25:38,194][metrics][INFO] - Evaluating wf_Q_A_Prob_normalised
 [2025-05-02 20:25:39,628][metrics][INFO] - Evaluating wf_Q_A_ROUGE
 [2025-05-02 20:26:00,244][metrics][INFO] - Skipping wf_Truth_Ratio's precompute wf_Q_A_Prob, already evaluated.
 [2025-05-02 20:26:00,244][metrics][INFO] - Skipping wf_Truth_Ratio's precompute wf_Q_A_PERT_Prob, already evaluated.
 [2025-05-02 20:26:00,244][metrics][INFO] - Evaluating wf_Truth_Ratio
 [2025-05-02 20:26:00,244][metrics][INFO] - Evaluating model_utility
 [2025-05-02 20:26:00,245][evaluator][INFO] - Result for metric model_utility:	0.5869124818129423
 [2025-05-02 20:26:03,404][metrics][INFO] - Loading evaluations from saves/eval/tofu_Llama-3.2-3B-Instruct_retain90/TOFU_EVAL.json
 [2025-05-02 20:26:03,421][metrics][INFO] - Evaluating mia_min_k
 [2025-05-02 20:26:12,296][metrics][INFO] - Loading evaluations from saves/eval/tofu_Llama-3.2-3B-Instruct_retain90/TOFU_EVAL.json
 [2025-05-02 20:26:12,305][metrics][INFO] - Evaluating privleak
 [2025-05-02 20:26:12,305][evaluator][INFO] - Result for metric privleak:	64.70836985162877
 [2025-05-02 20:26:14,371][metrics][INFO] - Evaluating extraction_strength
 [2025-05-02 20:26:19,115][evaluator][INFO] - Result for metric extraction_strength:	0.03250892997513522
 [2025-05-09 19:46:14,049][model][INFO] - Setting pad_token as eos token: <|eot_id|>
 [2025-05-09 19:46:14,052][evaluator][INFO] - Evaluations stored in the experiment directory: saves/unlearn/tofu_Llama-3.2-3B-Instruct_forget10_GradDiff/evals
 [2025-05-09 19:46:14,053][evaluator][INFO] - Loading existing evaluations from saves/unlearn/tofu_Llama-3.2-3B-Instruct_forget10_GradDiff/evals/TOFU_EVAL.json
 [2025-05-09 19:46:14,114][evaluator][INFO] - ***** Running TOFU evaluation suite *****
 [2025-05-09 19:46:14,114][evaluator][INFO] - Fine-grained evaluations will be saved to: saves/unlearn/tofu_Llama-3.2-3B-Instruct_forget10_GradDiff/evals/TOFU_EVAL.json
 [2025-05-09 19:46:14,114][evaluator][INFO] - Aggregated evaluations will be summarised in: saves/unlearn/tofu_Llama-3.2-3B-Instruct_forget10_GradDiff/evals/TOFU_SUMMARY.json
 [2025-05-09 19:46:14,114][evaluator][INFO] - Skipping forget_quality, already evaluated.
 [2025-05-09 19:46:14,114][evaluator][INFO] - Result for metric forget_quality:	2.50770024871112e-208
 [2025-05-09 19:46:14,127][evaluator][INFO] - Skipping forget_Q_A_Prob, already evaluated.
 [2025-05-09 19:46:14,127][evaluator][INFO] - Result for metric forget_Q_A_Prob:	4.883993844752212e-09
 [2025-05-09 19:46:14,132][evaluator][INFO] - Skipping forget_Q_A_ROUGE, already evaluated.
 [2025-05-09 19:46:14,132][evaluator][INFO] - Result for metric forget_Q_A_ROUGE:	0.005082328795915088
 [2025-05-09 19:46:14,165][evaluator][INFO] - Skipping model_utility, already evaluated.
 [2025-05-09 19:46:14,165][evaluator][INFO] - Result for metric model_utility:	0.5869124818129423
 [2025-05-09 19:46:14,170][evaluator][INFO] - Skipping privleak, already evaluated.
 [2025-05-09 19:46:14,170][evaluator][INFO] - Result for metric privleak:	64.70836985162877
 [2025-05-09 19:46:14,187][evaluator][INFO] - Skipping extraction_strength, already evaluated.
 [2025-05-09 19:46:14,188][evaluator][INFO] - Result for metric extraction_strength:	0.03250892997513522
 [2025-05-13 07:37:13,698][model][INFO] - Setting pad_token as eos token: <|eot_id|>
 [2025-05-13 07:37:13,701][evaluator][INFO] - Evaluations stored in the experiment directory: saves/unlearn/tofu_Llama-3.2-3B-Instruct_forget10_GradDiff/evals
 [2025-05-13 07:37:13,702][evaluator][INFO] - Loading existing evaluations from saves/unlearn/tofu_Llama-3.2-3B-Instruct_forget10_GradDiff/evals/TOFU_EVAL.json
 [2025-05-13 07:37:13,777][evaluator][INFO] - ***** Running TOFU evaluation suite *****
 [2025-05-13 07:37:13,777][evaluator][INFO] - Fine-grained evaluations will be saved to: saves/unlearn/tofu_Llama-3.2-3B-Instruct_forget10_GradDiff/evals/TOFU_EVAL.json
 [2025-05-13 07:37:13,778][evaluator][INFO] - Aggregated evaluations will be summarised in: saves/unlearn/tofu_Llama-3.2-3B-Instruct_forget10_GradDiff/evals/TOFU_SUMMARY.json
 [2025-05-13 07:37:13,778][evaluator][INFO] - Skipping forget_quality, already evaluated.
 [2025-05-13 07:37:13,778][evaluator][INFO] - Result for metric forget_quality:	2.50770024871112e-208
 [2025-05-13 07:37:13,807][evaluator][INFO] - Skipping forget_Q_A_Prob, already evaluated.
 [2025-05-13 07:37:13,807][evaluator][INFO] - Result for metric forget_Q_A_Prob:	4.883993844752212e-09
 [2025-05-13 07:37:13,811][evaluator][INFO] - Skipping forget_Q_A_ROUGE, already evaluated.
 [2025-05-13 07:37:13,811][evaluator][INFO] - Result for metric forget_Q_A_ROUGE:	0.005082328795915088
 [2025-05-13 07:37:13,813][evaluator][INFO] - Skipping model_utility, already evaluated.
 [2025-05-13 07:37:13,813][evaluator][INFO] - Result for metric model_utility:	0.5869124818129423
 [2025-05-13 07:37:13,815][evaluator][INFO] - Skipping privleak, already evaluated.
 [2025-05-13 07:37:13,815][evaluator][INFO] - Result for metric privleak:	64.70836985162877
 [2025-05-13 07:37:13,818][evaluator][INFO] - Skipping extraction_strength, already evaluated.
 [2025-05-13 07:37:13,818][evaluator][INFO] - Result for metric extraction_strength:	0.03250892997513522
--- a/generation_config.json
+++ b/generation_config.json
@@ -0,0 +1,12 @@
 {
  "bos_token_id": 128000,
  "do_sample": true,
  "eos_token_id": [
    128001,
    128008,
    128009
  ],
  "temperature": 0.6,
  "top_p": 0.9,
  "transformers_version": "4.51.3"
 }
--- a/logs/events.out.tfevents.1746183760.node3.2818538.0
+++ b/logs/events.out.tfevents.1746183760.node3.2818538.0
@@ -0,0 +1,3 @@
 version https://git-lfs.github.com/spec/v1
 oid sha256:61900df4734080da29174b259d940635394b2398d22479a39e13d2c6f2c3b5b3
 size 10647
--- a/logs/events.out.tfevents.1747087498.node4.1492387.0
+++ b/logs/events.out.tfevents.1747087498.node4.1492387.0
@@ -0,0 +1,3 @@
 version https://git-lfs.github.com/spec/v1
 oid sha256:65abc3507233ded20a7f552dc7b6ffcd75185c2a40ae74f076bae81d2537286c
 size 10616
--- a/model-00001-of-00002.safetensors
+++ b/model-00001-of-00002.safetensors
@@ -0,0 +1,3 @@
 version https://git-lfs.github.com/spec/v1
 oid sha256:8799cb14f6a62625168173237ee42a214e51421629ad6b771419f971c2ec054d
 size 4965799096
--- a/model-00002-of-00002.safetensors
+++ b/model-00002-of-00002.safetensors
@@ -0,0 +1,3 @@
 version https://git-lfs.github.com/spec/v1
 oid sha256:0e87588048e4b3e54043caf183c98509fb1dcbdb4b31cf058f1ee18492785e63
 size 1459729952
--- a/model.safetensors.index.json
+++ b/model.safetensors.index.json
@@ -0,0 +1,261 @@
 {
  "metadata": {
    "total_size": 6425499648
  },
  "weight_map": {
    "model.embed_tokens.weight": "model-00001-of-00002.safetensors",
    "model.layers.0.input_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.0.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.0.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.0.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.0.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.0.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.0.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.0.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.0.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.1.input_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.1.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.1.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.1.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.1.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.1.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.1.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.1.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.1.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.10.input_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.10.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.10.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.10.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.10.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.10.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.10.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.10.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.10.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.11.input_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.11.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.11.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.11.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.11.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.11.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.11.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.11.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.11.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.12.input_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.12.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.12.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.12.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.12.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.12.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.12.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.12.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.12.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.13.input_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.13.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.13.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.13.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.13.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.13.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.13.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.13.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.13.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.14.input_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.14.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.14.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.14.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.14.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.14.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.14.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.14.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.14.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.15.input_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.15.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.15.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.15.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.15.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.15.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.15.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.15.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.15.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.16.input_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.16.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.16.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.16.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.16.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.16.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.16.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.16.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.16.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.17.input_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.17.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.17.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.17.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.17.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.17.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.17.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.17.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.17.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.18.input_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.18.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.18.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.18.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.18.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.18.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.18.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.18.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.18.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.19.input_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.19.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.19.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.19.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.19.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.19.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.19.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.19.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.19.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.2.input_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.2.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.2.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.2.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.2.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.2.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.2.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.2.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.2.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.20.input_layernorm.weight": "model-00002-of-00002.safetensors",
    "model.layers.20.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.20.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.20.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.20.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
    "model.layers.20.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.20.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.20.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.20.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.21.input_layernorm.weight": "model-00002-of-00002.safetensors",
    "model.layers.21.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.21.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.21.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.21.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
    "model.layers.21.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.21.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.21.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.21.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.22.input_layernorm.weight": "model-00002-of-00002.safetensors",
    "model.layers.22.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.22.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.22.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.22.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
    "model.layers.22.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.22.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.22.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.22.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.23.input_layernorm.weight": "model-00002-of-00002.safetensors",
    "model.layers.23.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.23.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.23.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.23.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
    "model.layers.23.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.23.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.23.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.23.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.24.input_layernorm.weight": "model-00002-of-00002.safetensors",
    "model.layers.24.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.24.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.24.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.24.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
    "model.layers.24.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.24.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.24.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.24.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.25.input_layernorm.weight": "model-00002-of-00002.safetensors",
    "model.layers.25.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.25.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.25.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.25.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
    "model.layers.25.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.25.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.25.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.25.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.26.input_layernorm.weight": "model-00002-of-00002.safetensors",
    "model.layers.26.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.26.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.26.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.26.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
    "model.layers.26.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.26.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.26.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.26.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.27.input_layernorm.weight": "model-00002-of-00002.safetensors",
    "model.layers.27.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.27.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.27.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.27.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
    "model.layers.27.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.27.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.27.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.27.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
    "model.layers.3.input_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.3.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.3.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.3.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.3.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.3.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.3.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.3.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.3.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.4.input_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.4.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.4.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.4.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.4.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.4.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.4.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.4.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.4.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.5.input_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.5.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.5.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.5.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.5.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.5.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.5.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.5.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.5.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.6.input_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.6.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.6.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.6.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.6.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.6.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.6.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.6.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.6.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.7.input_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.7.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.7.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.7.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.7.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.7.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.7.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.7.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.7.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.8.input_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.8.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.8.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.8.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.8.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.8.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.8.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.8.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.8.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.9.input_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.9.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.9.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.9.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.9.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
    "model.layers.9.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.9.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.9.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
    "model.layers.9.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
    "model.norm.weight": "model-00002-of-00002.safetensors"
  }
 }
--- a/special_tokens_map.json
+++ b/special_tokens_map.json
@@ -0,0 +1,17 @@
 {
  "bos_token": {
    "content": "<|begin_of_text|>",
    "lstrip": false,
    "normalized": false,
    "rstrip": false,
    "single_word": false
  },
  "eos_token": {
    "content": "<|eot_id|>",
    "lstrip": false,
    "normalized": false,
    "rstrip": false,
    "single_word": false
  },
  "pad_token": "<|eot_id|>"
 }
--- a/tokenizer.json
+++ b/tokenizer.json
--- a/tokenizer_config.json
+++ b/tokenizer_config.json
--- a/trainer_state.json
+++ b/trainer_state.json
@@ -0,0 +1,211 @@
 {
  "best_global_step": null,
  "best_metric": null,
  "best_model_checkpoint": null,
  "epoch": 9.24,
  "eval_steps": 500,
  "global_step": 120,
  "is_hyper_param_search": false,
  "is_local_process_zero": true,
  "is_world_process_zero": true,
  "log_history": [
    {
      "epoch": 0.4,
      "grad_norm": 18.6333627466956,
      "learning_rate": 3.3333333333333333e-06,
      "loss": -0.028,
      "step": 5
    },
    {
      "epoch": 0.8,
      "grad_norm": 16.032357759228358,
      "learning_rate": 7.500000000000001e-06,
      "loss": -0.064,
      "step": 10
    },
    {
      "epoch": 1.16,
      "grad_norm": 34.99217902500999,
      "learning_rate": 9.814814814814815e-06,
      "loss": -0.3462,
      "step": 15
    },
    {
      "epoch": 1.56,
      "grad_norm": 247.49125320648662,
      "learning_rate": 9.351851851851854e-06,
      "loss": -1.9847,
      "step": 20
    },
    {
      "epoch": 1.96,
      "grad_norm": 1038.7636425274402,
      "learning_rate": 8.888888888888888e-06,
      "loss": -7.2508,
      "step": 25
    },
    {
      "epoch": 2.32,
      "grad_norm": 3263.564556779214,
      "learning_rate": 8.425925925925926e-06,
      "loss": -12.6012,
      "step": 30
    },
    {
      "epoch": 2.7199999999999998,
      "grad_norm": 5369.050190792846,
      "learning_rate": 7.962962962962963e-06,
      "loss": -85.1783,
      "step": 35
    },
    {
      "epoch": 3.08,
      "grad_norm": 3475.076920659395,
      "learning_rate": 7.500000000000001e-06,
      "loss": -229.2591,
      "step": 40
    },
    {
      "epoch": 3.48,
      "grad_norm": 3651.1737873444736,
      "learning_rate": 7.0370370370370375e-06,
      "loss": -356.1301,
      "step": 45
    },
    {
      "epoch": 3.88,
      "grad_norm": 439.7775722171736,
      "learning_rate": 6.574074074074075e-06,
      "loss": -378.3872,
      "step": 50
    },
    {
      "epoch": 4.24,
      "grad_norm": 434.2588747466371,
      "learning_rate": 6.111111111111112e-06,
      "loss": -341.9133,
      "step": 55
    },
    {
      "epoch": 4.64,
      "grad_norm": 444.98196659861003,
      "learning_rate": 5.6481481481481485e-06,
      "loss": -378.8201,
      "step": 60
    },
    {
      "epoch": 5.0,
      "grad_norm": 408.67513933006654,
      "learning_rate": 5.185185185185185e-06,
      "loss": -345.2147,
      "step": 65
    },
    {
      "epoch": 5.4,
      "grad_norm": 392.3829022170684,
      "learning_rate": 4.722222222222222e-06,
      "loss": -385.2472,
      "step": 70
    },
    {
      "epoch": 5.8,
      "grad_norm": 372.4425529988918,
      "learning_rate": 4.2592592592592596e-06,
      "loss": -384.5583,
      "step": 75
    },
    {
      "epoch": 6.16,
      "grad_norm": 376.5960722072173,
      "learning_rate": 3.796296296296297e-06,
      "loss": -356.0032,
      "step": 80
    },
    {
      "epoch": 6.5600000000000005,
      "grad_norm": 395.12972202440875,
      "learning_rate": 3.3333333333333333e-06,
      "loss": -388.6089,
      "step": 85
    },
    {
      "epoch": 6.96,
      "grad_norm": 367.0863836880796,
      "learning_rate": 2.8703703703703706e-06,
      "loss": -389.7163,
      "step": 90
    },
    {
      "epoch": 7.32,
      "grad_norm": 519.4771966031532,
      "learning_rate": 2.4074074074074075e-06,
      "loss": -356.6014,
      "step": 95
    },
    {
      "epoch": 7.72,
      "grad_norm": 378.1832167305145,
      "learning_rate": 1.944444444444445e-06,
      "loss": -392.9878,
      "step": 100
    },
    {
      "epoch": 8.08,
      "grad_norm": 381.25866581784237,
      "learning_rate": 1.4814814814814815e-06,
      "loss": -356.5252,
      "step": 105
    },
    {
      "epoch": 8.48,
      "grad_norm": 1027.7502523973324,
      "learning_rate": 1.0185185185185185e-06,
      "loss": -395.6457,
      "step": 110
    },
    {
      "epoch": 8.88,
      "grad_norm": 502.8117462390182,
      "learning_rate": 5.555555555555555e-07,
      "loss": -396.9123,
      "step": 115
    },
    {
      "epoch": 9.24,
      "grad_norm": 914.2328850512966,
      "learning_rate": 9.259259259259259e-08,
      "loss": -359.5569,
      "step": 120
    },
    {
      "epoch": 9.24,
      "step": 120,
      "total_flos": 0.0,
      "train_loss": -262.48087577770156,
      "train_runtime": 1884.8014,
      "train_samples_per_second": 2.122,
      "train_steps_per_second": 0.064
    }
  ],
  "logging_steps": 5,
  "max_steps": 120,
  "num_input_tokens_seen": 0,
  "num_train_epochs": 10,
  "save_steps": 500,
  "stateful_callbacks": {
    "TrainerControl": {
      "args": {
        "should_epoch_stop": false,
        "should_evaluate": false,
        "should_log": false,
        "should_save": false,
        "should_training_stop": false
      },
      "attributes": {}
    }
  },
  "total_flos": 0.0,
  "train_batch_size": 4,
  "trial_name": null,
  "trial_params": null
 }
--- a/training_args.bin
+++ b/training_args.bin
@@ -0,0 +1,3 @@
 version https://git-lfs.github.com/spec/v1
 oid sha256:d4c553d2bc4d5ea801499793514f69e838189da2748c5126ed5f6cb0c6dda127
 size 6968