commit ca60016a558a605028476bfd2556128d8ff322bd Author: ModelHub XC Date: Sun Sep 13 17:18:17 2026 +0800 初始化项目,由ModelHub XC社区提供模型 Model: adityabanerjee13/qwen2.5-0.5b-sft-IT Source: Original Platform diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..52373fe --- /dev/null +++ b/.gitattributes @@ -0,0 +1,36 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text +tokenizer.json filter=lfs diff=lfs merge=lfs -text diff --git a/README.md b/README.md new file mode 100644 index 0000000..8ae3d41 --- /dev/null +++ b/README.md @@ -0,0 +1,179 @@ +--- +library_name: transformers +license: apache-2.0 +base_model: adityabanerjee13/qwen2.5-0.5b-cpt-mix-1to2 +tags: +- axolotl +- generated_from_trainer +datasets: +- adityabanerjee13/indic-sft-mini-train +- adityabanerjee13/tulu-sft-mini-train +model-index: +- name: qwen2.5-0.5b-sft-IT + results: [] +--- + + + +[Built with Axolotl](https://github.com/axolotl-ai-cloud/axolotl) +
See axolotl config + +axolotl version: `0.19.0.dev0` +```yaml +# ============================================================================== +# Axolotl CPT config — Qwen2.5-0.5B, full-parameter, single GPU. +# Data mix RATIO EXPERIMENT (character-level exact): +# +# RUN 2 of 3 — fineweb : indic = 1 : 2 (FineWeb is HALF the Indic size) +# FineWeb web-crawl chars == Indic train chars / 2. +# +# Indic train : adityabanerjee13/indic-cpt-mini-train (7,907,882 chars) +# FineWeb train: adityabanerjee13/fineweb-cpt-half (3,953,941 chars) +# Validation : adityabanerjee13/indic-cpt-mini-val (held-out 1% Indic) +# +# The datasets are pre-sized to exact character counts on the Hub, so loading +# each one whole gives the exact 1:2 ratio — no slicing needed. +# +# Usage: +# python train.py --config qwen2.5_0.5b_cpt_mix_1to2.yml +# ============================================================================== + +base_model: adityabanerjee13/qwen2.5-0.5b-cpt-mix-1to2 +model_type: AutoModelForCausalLM +tokenizer_type: AutoTokenizer +trust_remote_code: false + +adapter: +load_in_8bit: false +load_in_4bit: false + +# multi_eval_plugin splits test_datasets back into per-source eval sets so +# this run logs eval_indic_cpt_mini_val_loss and eval_fineweb_cpt_val_loss +# separately (instead of one merged eval_loss) at every eval step, incl. to +# wandb. Requires this folder on PYTHONPATH — launch via +# `python train.py --config `. +plugins: + - multi_eval_plugin.MultiEvalPlugin + +datasets: + - path: adityabanerjee13/indic-sft-mini-train + type: chat_template + field_messages: messages + split: train + - path: adityabanerjee13/tulu-sft-mini-train + type: chat_template + field_messages: messages + split: train + +test_datasets: + - path: adityabanerjee13/indic-sft-mini-val + type: chat_template + field_messages: messages + split: validation + - path: adityabanerjee13/tulu-sft-mini-val + type: chat_template + field_messages: messages + split: validation + +train_on_inputs: false + +chat_template: tokenizer_default + +dataset_prepared_path: ./last_run_prepared_1to2 +dataset_num_proc: 1 # single-process tokenize: avoids fork deadlock +val_set_size: 0 +output_dir: ./outputs/qwen2.5-0.5b-sft-IT + +# --- Sequence packing ----------------------------------------------------- +sequence_len: 4096 +sample_packing: true +pad_to_sequence_len: true +eval_sample_packing: false + +# --- Optimization ---------------------------------------------------------- +gradient_accumulation_steps: 8 +micro_batch_size: 4 +num_epochs: 3 +optimizer: adamw_torch_fused +lr_scheduler: cosine +learning_rate: 2e-5 +warmup_ratio: 0.03 +weight_decay: 0.01 +max_grad_norm: 1.0 + +train_on_inputs: true +group_by_length: false + +# --- Precision / memory --------------------------------------------------- +bf16: auto +fp16: +tf32: true +gradient_checkpointing: true +flash_attention: true + +# --- Logging / checkpoints ------------------------------------------------ +logging_steps: 10 +save_strategy: steps +save_steps: 500 +save_total_limit: 30 +save_only_model: true # save weights only — no optimizer/scheduler state + # (checkpoints ~1/3 the size; can't resume training) +evals_per_epoch: 4 + +wandb_project: indic-sft +wandb_entity: models-na9841 +wandb_name: qwen2.5-0.5b-sft-IT +wandb_log_model: "false" + +hub_model_id: adityabanerjee13/qwen2.5-0.5b-sft-IT +hub_strategy: all_checkpoints + +special_tokens: + +``` + +

+ +# qwen2.5-0.5b-sft-IT + +This model is a fine-tuned version of [adityabanerjee13/qwen2.5-0.5b-cpt-mix-1to2](https://huggingface.co/adityabanerjee13/qwen2.5-0.5b-cpt-mix-1to2) on the adityabanerjee13/indic-sft-mini-train and the adityabanerjee13/tulu-sft-mini-train datasets. + +## Model description + +More information needed + +## Intended uses & limitations + +More information needed + +## Training and evaluation data + +More information needed + +## Training procedure + +### Training hyperparameters + +The following hyperparameters were used during training: +- learning_rate: 2e-05 +- train_batch_size: 4 +- eval_batch_size: 4 +- seed: 42 +- gradient_accumulation_steps: 8 +- total_train_batch_size: 32 +- optimizer: Use OptimizerNames.ADAMW_TORCH_FUSED with betas=(0.9,0.999) and epsilon=1e-08 and optimizer_args=No additional optimizer arguments +- lr_scheduler_type: cosine +- lr_scheduler_warmup_steps: 17 +- training_steps: 591 + +### Training results + + + +### Framework versions + +- Transformers 5.14.1 +- Pytorch 2.12.0+cu130 +- Datasets 4.8.4 +- Tokenizers 0.22.2 diff --git a/chat_template.jinja b/chat_template.jinja new file mode 100644 index 0000000..28028c0 --- /dev/null +++ b/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-500/chat_template.jinja b/checkpoint-500/chat_template.jinja new file mode 100644 index 0000000..28028c0 --- /dev/null +++ b/checkpoint-500/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-500/config.json b/checkpoint-500/config.json new file mode 100644 index 0000000..cf68265 --- /dev/null +++ b/checkpoint-500/config.json @@ -0,0 +1,58 @@ +{ + "architectures": [ + "Qwen2ForCausalLM" + ], + "attention_dropout": 0.0, + "bos_token_id": null, + "dtype": "bfloat16", + "eos_token_id": 151643, + "hidden_act": "silu", + "hidden_size": 896, + "initializer_range": 0.02, + "intermediate_size": 4864, + "layer_types": [ + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention" + ], + "max_position_embeddings": 32768, + "max_window_layers": 24, + "model_type": "qwen2", + "num_attention_heads": 14, + "num_hidden_layers": 24, + "num_key_value_heads": 2, + "pad_token_id": 151643, + "rms_norm_eps": 1e-06, + "rope_parameters": { + "rope_theta": 1000000.0, + "rope_type": "default" + }, + "sliding_window": null, + "tie_word_embeddings": true, + "transformers_version": "5.14.1", + "use_cache": false, + "use_mrope": false, + "use_sliding_window": false, + "vocab_size": 151936 +} diff --git a/checkpoint-500/generation_config.json b/checkpoint-500/generation_config.json new file mode 100644 index 0000000..c2d1ff4 --- /dev/null +++ b/checkpoint-500/generation_config.json @@ -0,0 +1,9 @@ +{ + "do_sample": true, + "eos_token_id": [ + 151643 + ], + "max_new_tokens": 2048, + "pad_token_id": 151643, + "transformers_version": "5.14.1" +} diff --git a/checkpoint-500/model.safetensors b/checkpoint-500/model.safetensors new file mode 100644 index 0000000..00857f5 --- /dev/null +++ b/checkpoint-500/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9df9ccbbb1556eb3db2d7023d7cef1f34ce7492419e30bad0da7dcb729885eab +size 988097824 diff --git a/checkpoint-500/tokenizer.json b/checkpoint-500/tokenizer.json new file mode 100644 index 0000000..34510ff --- /dev/null +++ b/checkpoint-500/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3fd169731d2cbde95e10bf356d66d5997fd885dd8dbb6fb4684da3f23b2585d8 +size 11421892 diff --git a/checkpoint-500/tokenizer_config.json b/checkpoint-500/tokenizer_config.json new file mode 100644 index 0000000..df536e9 --- /dev/null +++ b/checkpoint-500/tokenizer_config.json @@ -0,0 +1,30 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "local_files_only": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-500/tokens_state.json b/checkpoint-500/tokens_state.json new file mode 100644 index 0000000..ec9fcd0 --- /dev/null +++ b/checkpoint-500/tokens_state.json @@ -0,0 +1 @@ +{"total": 65380352, "trainable": 63254896} \ No newline at end of file diff --git a/checkpoint-500/trainer_state.json b/checkpoint-500/trainer_state.json new file mode 100644 index 0000000..331da9f --- /dev/null +++ b/checkpoint-500/trainer_state.json @@ -0,0 +1,975 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 2.526582278481013, + "eval_steps": 50, + "global_step": 500, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0, + "eval_indic_sft_mini_val_loss": 1.127184510231018, + "eval_indic_sft_mini_val_runtime": 156.0033, + "eval_indic_sft_mini_val_samples_per_second": 12.307, + "eval_indic_sft_mini_val_steps_per_second": 3.077, + "memory/device_reserved (GiB)": 24.22, + "memory/max_active (GiB)": 24.14, + "memory/max_allocated (GiB)": 24.14, + "step": 0 + }, + { + "epoch": 0, + "eval_tulu_sft_mini_val_loss": 2.330348491668701, + "eval_tulu_sft_mini_val_runtime": 91.8069, + "eval_tulu_sft_mini_val_samples_per_second": 12.548, + "eval_tulu_sft_mini_val_steps_per_second": 3.137, + "memory/device_reserved (GiB)": 24.22, + "memory/max_active (GiB)": 24.14, + "memory/max_allocated (GiB)": 24.14, + "step": 0 + }, + { + "epoch": 0.05063291139240506, + "grad_norm": 1.9375, + "learning_rate": 1.0588235294117648e-05, + "loss": 1.2239381790161132, + "memory/device_reserved (GiB)": 37.02, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 3.40055, + "step": 10, + "tokens/total": 1310720, + "tokens/trainable": 1268231 + }, + { + "epoch": 0.10126582278481013, + "grad_norm": 1.640625, + "learning_rate": 1.9999400896826965e-05, + "loss": 1.1529170989990234, + "memory/device_reserved (GiB)": 37.02, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 3.16742, + "step": 20, + "tokens/total": 2621440, + "tokens/train_per_sec_per_gpu": 17988.68, + "tokens/trainable": 2535045 + }, + { + "epoch": 0.1518987341772152, + "grad_norm": 1.4453125, + "learning_rate": 1.9978439822224228e-05, + "loss": 1.1212153434753418, + "memory/device_reserved (GiB)": 37.02, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 3.06858, + "step": 30, + "tokens/total": 3932160, + "tokens/train_per_sec_per_gpu": 17958.15, + "tokens/trainable": 3804033 + }, + { + "epoch": 0.20253164556962025, + "grad_norm": 1.3828125, + "learning_rate": 1.9927595335238736e-05, + "loss": 1.106326198577881, + "memory/device_reserved (GiB)": 37.02, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 3.02323, + "step": 40, + "tokens/total": 5242880, + "tokens/train_per_sec_per_gpu": 17929.18, + "tokens/trainable": 5070883 + }, + { + "epoch": 0.25316455696202533, + "grad_norm": 1.421875, + "learning_rate": 1.984701970484229e-05, + "loss": 1.0841044425964355, + "memory/device_reserved (GiB)": 37.02, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.95679, + "step": 50, + "tokens/total": 6553600, + "tokens/train_per_sec_per_gpu": 17943.82, + "tokens/trainable": 6341109 + }, + { + "epoch": 0.25316455696202533, + "eval_indic_sft_mini_val_loss": 1.0926785469055176, + "eval_indic_sft_mini_val_runtime": 157.8472, + "eval_indic_sft_mini_val_samples_per_second": 12.164, + "eval_indic_sft_mini_val_steps_per_second": 3.041, + "memory/device_reserved (GiB)": 37.02, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 50 + }, + { + "epoch": 0.25316455696202533, + "eval_tulu_sft_mini_val_loss": 2.3083696365356445, + "eval_tulu_sft_mini_val_runtime": 92.1987, + "eval_tulu_sft_mini_val_samples_per_second": 12.495, + "eval_tulu_sft_mini_val_steps_per_second": 3.124, + "memory/device_reserved (GiB)": 26.58, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 50 + }, + { + "epoch": 0.3037974683544304, + "grad_norm": 1.34375, + "learning_rate": 1.9736954238777793e-05, + "loss": 1.1303999900817872, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 3.09689, + "step": 60, + "tokens/total": 7864320, + "tokens/train_per_sec_per_gpu": 3953.8, + "tokens/trainable": 7607732 + }, + { + "epoch": 0.35443037974683544, + "grad_norm": 1.4296875, + "learning_rate": 1.9597728560891266e-05, + "loss": 1.1227096557617187, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 3.07317, + "step": 70, + "tokens/total": 9175040, + "tokens/train_per_sec_per_gpu": 17981.63, + "tokens/trainable": 8877038 + }, + { + "epoch": 0.4050632911392405, + "grad_norm": 1.3828125, + "learning_rate": 1.9429759623974992e-05, + "loss": 1.109241771697998, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 3.03206, + "step": 80, + "tokens/total": 10485760, + "tokens/train_per_sec_per_gpu": 17934.18, + "tokens/trainable": 10145302 + }, + { + "epoch": 0.45569620253164556, + "grad_norm": 1.2578125, + "learning_rate": 1.9233550461078114e-05, + "loss": 1.071034049987793, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.9184, + "step": 90, + "tokens/total": 11796480, + "tokens/train_per_sec_per_gpu": 17952.48, + "tokens/trainable": 11415878 + }, + { + "epoch": 0.5063291139240507, + "grad_norm": 1.2421875, + "learning_rate": 1.900968867902419e-05, + "loss": 1.0816166877746582, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.94944, + "step": 100, + "tokens/total": 13107200, + "tokens/train_per_sec_per_gpu": 17943.02, + "tokens/trainable": 12684573 + }, + { + "epoch": 0.5063291139240507, + "eval_indic_sft_mini_val_loss": 1.0623797178268433, + "eval_indic_sft_mini_val_runtime": 157.7495, + "eval_indic_sft_mini_val_samples_per_second": 12.171, + "eval_indic_sft_mini_val_steps_per_second": 3.043, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 100 + }, + { + "epoch": 0.5063291139240507, + "eval_tulu_sft_mini_val_loss": 2.2967746257781982, + "eval_tulu_sft_mini_val_runtime": 92.342, + "eval_tulu_sft_mini_val_samples_per_second": 12.475, + "eval_tulu_sft_mini_val_steps_per_second": 3.119, + "memory/device_reserved (GiB)": 26.58, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 100 + }, + { + "epoch": 0.5569620253164557, + "grad_norm": 1.2421875, + "learning_rate": 1.8758844698647457e-05, + "loss": 1.106839370727539, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 3.02478, + "step": 110, + "tokens/total": 14417920, + "tokens/train_per_sec_per_gpu": 3955.64, + "tokens/trainable": 13951995 + }, + { + "epoch": 0.6075949367088608, + "grad_norm": 1.8984375, + "learning_rate": 1.848176974701775e-05, + "loss": 1.0786801338195802, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.9408, + "step": 120, + "tokens/total": 15728640, + "tokens/train_per_sec_per_gpu": 17975.05, + "tokens/trainable": 15221447 + }, + { + "epoch": 0.6582278481012658, + "grad_norm": 1.28125, + "learning_rate": 1.8179293607667177e-05, + "loss": 1.0739567756652832, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.92694, + "step": 130, + "tokens/total": 17039360, + "tokens/train_per_sec_per_gpu": 17954.31, + "tokens/trainable": 16490903 + }, + { + "epoch": 0.7088607594936709, + "grad_norm": 1.2890625, + "learning_rate": 1.7852322135555946e-05, + "loss": 1.0759190559387206, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.93269, + "step": 140, + "tokens/total": 18350080, + "tokens/train_per_sec_per_gpu": 17887.36, + "tokens/trainable": 17757184 + }, + { + "epoch": 0.759493670886076, + "grad_norm": 1.3828125, + "learning_rate": 1.7501834544219697e-05, + "loss": 1.0366504669189454, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.81976, + "step": 150, + "tokens/total": 19660800, + "tokens/train_per_sec_per_gpu": 17911.77, + "tokens/trainable": 19024962 + }, + { + "epoch": 0.759493670886076, + "eval_indic_sft_mini_val_loss": 1.0453351736068726, + "eval_indic_sft_mini_val_runtime": 157.7623, + "eval_indic_sft_mini_val_samples_per_second": 12.17, + "eval_indic_sft_mini_val_steps_per_second": 3.043, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 150 + }, + { + "epoch": 0.759493670886076, + "eval_tulu_sft_mini_val_loss": 2.2883615493774414, + "eval_tulu_sft_mini_val_runtime": 92.4616, + "eval_tulu_sft_mini_val_samples_per_second": 12.459, + "eval_tulu_sft_mini_val_steps_per_second": 3.115, + "memory/device_reserved (GiB)": 26.58, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 150 + }, + { + "epoch": 0.810126582278481, + "grad_norm": 1.328125, + "learning_rate": 1.7128880473222688e-05, + "loss": 1.0812637329101562, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.9484, + "step": 160, + "tokens/total": 20971520, + "tokens/train_per_sec_per_gpu": 3951.59, + "tokens/trainable": 20291324 + }, + { + "epoch": 0.8607594936708861, + "grad_norm": 1.328125, + "learning_rate": 1.6734576844699234e-05, + "loss": 1.0667606353759767, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.90595, + "step": 170, + "tokens/total": 22282240, + "tokens/train_per_sec_per_gpu": 17990.97, + "tokens/trainable": 21559984 + }, + { + "epoch": 0.9113924050632911, + "grad_norm": 1.3515625, + "learning_rate": 1.6320104518397473e-05, + "loss": 1.067934513092041, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.90936, + "step": 180, + "tokens/total": 23592960, + "tokens/train_per_sec_per_gpu": 17944.73, + "tokens/trainable": 22828372 + }, + { + "epoch": 0.9620253164556962, + "grad_norm": 1.296875, + "learning_rate": 1.588670475524283e-05, + "loss": 1.04850492477417, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.85338, + "step": 190, + "tokens/total": 24903680, + "tokens/train_per_sec_per_gpu": 17890.69, + "tokens/trainable": 24094636 + }, + { + "epoch": 1.010126582278481, + "grad_norm": 1.3046875, + "learning_rate": 1.5435675500012212e-05, + "loss": 1.0491924285888672, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.85534, + "step": 200, + "tokens/total": 26136576, + "tokens/train_per_sec_per_gpu": 17465.18, + "tokens/trainable": 25286404 + }, + { + "epoch": 1.010126582278481, + "eval_indic_sft_mini_val_loss": 1.0354256629943848, + "eval_indic_sft_mini_val_runtime": 157.741, + "eval_indic_sft_mini_val_samples_per_second": 12.172, + "eval_indic_sft_mini_val_steps_per_second": 3.043, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 200 + }, + { + "epoch": 1.010126582278481, + "eval_tulu_sft_mini_val_loss": 2.291062593460083, + "eval_tulu_sft_mini_val_runtime": 92.344, + "eval_tulu_sft_mini_val_samples_per_second": 12.475, + "eval_tulu_sft_mini_val_steps_per_second": 3.119, + "memory/device_reserved (GiB)": 26.58, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 200 + }, + { + "epoch": 1.0607594936708862, + "grad_norm": 1.375, + "learning_rate": 1.4968367494251486e-05, + "loss": 1.0487144470214844, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.85398, + "step": 210, + "tokens/total": 27447296, + "tokens/train_per_sec_per_gpu": 3957.26, + "tokens/trainable": 26554084 + }, + { + "epoch": 1.111392405063291, + "grad_norm": 1.3671875, + "learning_rate": 1.4486180231077278e-05, + "loss": 1.0355692863464356, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.81671, + "step": 220, + "tokens/total": 28758016, + "tokens/train_per_sec_per_gpu": 17979.8, + "tokens/trainable": 27823280 + }, + { + "epoch": 1.1620253164556962, + "grad_norm": 1.265625, + "learning_rate": 1.3990557763977694e-05, + "loss": 1.0380287170410156, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.82365, + "step": 230, + "tokens/total": 30068736, + "tokens/train_per_sec_per_gpu": 17964.75, + "tokens/trainable": 29092706 + }, + { + "epoch": 1.2126582278481013, + "grad_norm": 1.2734375, + "learning_rate": 1.3482984382163713e-05, + "loss": 1.036821174621582, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.82024, + "step": 240, + "tokens/total": 31379456, + "tokens/train_per_sec_per_gpu": 17930.26, + "tokens/trainable": 30360732 + }, + { + "epoch": 1.2632911392405064, + "grad_norm": 1.2265625, + "learning_rate": 1.2964980165422701e-05, + "loss": 0.995778751373291, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.70683, + "step": 250, + "tokens/total": 32690176, + "tokens/train_per_sec_per_gpu": 17927.94, + "tokens/trainable": 31631352 + }, + { + "epoch": 1.2632911392405064, + "eval_indic_sft_mini_val_loss": 1.027944803237915, + "eval_indic_sft_mini_val_runtime": 157.8775, + "eval_indic_sft_mini_val_samples_per_second": 12.161, + "eval_indic_sft_mini_val_steps_per_second": 3.04, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 250 + }, + { + "epoch": 1.2632911392405064, + "eval_tulu_sft_mini_val_loss": 2.2984113693237305, + "eval_tulu_sft_mini_val_runtime": 92.3048, + "eval_tulu_sft_mini_val_samples_per_second": 12.48, + "eval_tulu_sft_mini_val_steps_per_second": 3.12, + "memory/device_reserved (GiB)": 26.58, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 250 + }, + { + "epoch": 1.3139240506329113, + "grad_norm": 1.2109375, + "learning_rate": 1.2438096431786408e-05, + "loss": 1.0438777923583984, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.84021, + "step": 260, + "tokens/total": 34000896, + "tokens/train_per_sec_per_gpu": 3955.89, + "tokens/trainable": 32899468 + }, + { + "epoch": 1.3645569620253164, + "grad_norm": 1.28125, + "learning_rate": 1.1903911091646684e-05, + "loss": 1.0240283012390137, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.78439, + "step": 270, + "tokens/total": 35311616, + "tokens/train_per_sec_per_gpu": 17961.08, + "tokens/trainable": 34168384 + }, + { + "epoch": 1.4151898734177215, + "grad_norm": 1.234375, + "learning_rate": 1.1364023922232503e-05, + "loss": 1.031261920928955, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.8046, + "step": 280, + "tokens/total": 36622336, + "tokens/train_per_sec_per_gpu": 17939.56, + "tokens/trainable": 35436704 + }, + { + "epoch": 1.4658227848101266, + "grad_norm": 1.3359375, + "learning_rate": 1.0820051776600175e-05, + "loss": 1.0369884490966796, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.82071, + "step": 290, + "tokens/total": 37933056, + "tokens/train_per_sec_per_gpu": 17936.98, + "tokens/trainable": 36705368 + }, + { + "epoch": 1.5164556962025317, + "grad_norm": 1.1875, + "learning_rate": 1.0273623741484924e-05, + "loss": 1.0238310813903808, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.78384, + "step": 300, + "tokens/total": 39243776, + "tokens/train_per_sec_per_gpu": 17926.64, + "tokens/trainable": 37973400 + }, + { + "epoch": 1.5164556962025317, + "eval_indic_sft_mini_val_loss": 1.0224663019180298, + "eval_indic_sft_mini_val_runtime": 157.771, + "eval_indic_sft_mini_val_samples_per_second": 12.17, + "eval_indic_sft_mini_val_steps_per_second": 3.042, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 300 + }, + { + "epoch": 1.5164556962025317, + "eval_tulu_sft_mini_val_loss": 2.295070171356201, + "eval_tulu_sft_mini_val_runtime": 92.349, + "eval_tulu_sft_mini_val_samples_per_second": 12.474, + "eval_tulu_sft_mini_val_steps_per_second": 3.119, + "memory/device_reserved (GiB)": 26.58, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 300 + }, + { + "epoch": 1.5670886075949366, + "grad_norm": 1.2421875, + "learning_rate": 9.726376258515077e-06, + "loss": 1.0644823074340821, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.89934, + "step": 310, + "tokens/total": 40554496, + "tokens/train_per_sec_per_gpu": 3956.31, + "tokens/trainable": 39240980 + }, + { + "epoch": 1.6177215189873417, + "grad_norm": 1.234375, + "learning_rate": 9.179948223399828e-06, + "loss": 1.0305088996887206, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.80249, + "step": 320, + "tokens/total": 41865216, + "tokens/train_per_sec_per_gpu": 17965.06, + "tokens/trainable": 40509632 + }, + { + "epoch": 1.6683544303797468, + "grad_norm": 1.265625, + "learning_rate": 8.6359760777675e-06, + "loss": 1.0250298500061035, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.78718, + "step": 330, + "tokens/total": 43175936, + "tokens/train_per_sec_per_gpu": 17953.88, + "tokens/trainable": 41777768 + }, + { + "epoch": 1.7189873417721517, + "grad_norm": 1.2265625, + "learning_rate": 8.096088908353316e-06, + "loss": 1.0106207847595214, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.74731, + "step": 340, + "tokens/total": 44486656, + "tokens/train_per_sec_per_gpu": 17933.81, + "tokens/trainable": 43045456 + }, + { + "epoch": 1.769620253164557, + "grad_norm": 1.2578125, + "learning_rate": 7.561903568213595e-06, + "loss": 0.9956131935119629, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.70638, + "step": 350, + "tokens/total": 45797376, + "tokens/train_per_sec_per_gpu": 17927.21, + "tokens/trainable": 44311528 + }, + { + "epoch": 1.769620253164557, + "eval_indic_sft_mini_val_loss": 1.0202709436416626, + "eval_indic_sft_mini_val_runtime": 157.7597, + "eval_indic_sft_mini_val_samples_per_second": 12.17, + "eval_indic_sft_mini_val_steps_per_second": 3.043, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 350 + }, + { + "epoch": 1.769620253164557, + "eval_tulu_sft_mini_val_loss": 2.2989728450775146, + "eval_tulu_sft_mini_val_runtime": 92.4784, + "eval_tulu_sft_mini_val_samples_per_second": 12.457, + "eval_tulu_sft_mini_val_steps_per_second": 3.114, + "memory/device_reserved (GiB)": 26.58, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 350 + }, + { + "epoch": 1.820253164556962, + "grad_norm": 1.2734375, + "learning_rate": 7.035019834577301e-06, + "loss": 1.0437658309936524, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.83989, + "step": 360, + "tokens/total": 47108096, + "tokens/train_per_sec_per_gpu": 3952.58, + "tokens/trainable": 45578288 + }, + { + "epoch": 1.870886075949367, + "grad_norm": 1.2265625, + "learning_rate": 6.517015617836292e-06, + "loss": 1.0040513038635255, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.72932, + "step": 370, + "tokens/total": 48418816, + "tokens/train_per_sec_per_gpu": 17939.3, + "tokens/trainable": 46846324 + }, + { + "epoch": 1.9215189873417722, + "grad_norm": 1.234375, + "learning_rate": 6.009442236022307e-06, + "loss": 1.0151902198791505, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.75989, + "step": 380, + "tokens/total": 49729536, + "tokens/train_per_sec_per_gpu": 17915.04, + "tokens/trainable": 48113428 + }, + { + "epoch": 1.972151898734177, + "grad_norm": 1.3125, + "learning_rate": 5.513819768922723e-06, + "loss": 0.997506046295166, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.71151, + "step": 390, + "tokens/total": 51040256, + "tokens/train_per_sec_per_gpu": 17963.73, + "tokens/trainable": 49382776 + }, + { + "epoch": 2.020253164556962, + "grad_norm": 1.2421875, + "learning_rate": 5.031632505748516e-06, + "loss": 1.0011167526245117, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.72132, + "step": 400, + "tokens/total": 52273152, + "tokens/train_per_sec_per_gpu": 17458.66, + "tokens/trainable": 50572704 + }, + { + "epoch": 2.020253164556962, + "eval_indic_sft_mini_val_loss": 1.0187087059020996, + "eval_indic_sft_mini_val_runtime": 158.0242, + "eval_indic_sft_mini_val_samples_per_second": 12.15, + "eval_indic_sft_mini_val_steps_per_second": 3.038, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 400 + }, + { + "epoch": 2.020253164556962, + "eval_tulu_sft_mini_val_loss": 2.2966930866241455, + "eval_tulu_sft_mini_val_runtime": 92.3222, + "eval_tulu_sft_mini_val_samples_per_second": 12.478, + "eval_tulu_sft_mini_val_steps_per_second": 3.12, + "memory/device_reserved (GiB)": 26.58, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 400 + }, + { + "epoch": 2.070886075949367, + "grad_norm": 1.2421875, + "learning_rate": 4.56432449998779e-06, + "loss": 1.0103323936462403, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.74651, + "step": 410, + "tokens/total": 53583872, + "tokens/train_per_sec_per_gpu": 3952.63, + "tokens/trainable": 51839796 + }, + { + "epoch": 2.1215189873417724, + "grad_norm": 1.234375, + "learning_rate": 4.113295244757171e-06, + "loss": 1.0133058547973632, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.75469, + "step": 420, + "tokens/total": 54894592, + "tokens/train_per_sec_per_gpu": 17984.79, + "tokens/trainable": 53108816 + }, + { + "epoch": 2.1721518987341772, + "grad_norm": 1.21875, + "learning_rate": 3.679895481602529e-06, + "loss": 1.0214984893798829, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.77735, + "step": 430, + "tokens/total": 56205312, + "tokens/train_per_sec_per_gpu": 17959.57, + "tokens/trainable": 54378168 + }, + { + "epoch": 2.222784810126582, + "grad_norm": 1.2109375, + "learning_rate": 3.2654231553007665e-06, + "loss": 0.9840593338012695, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.67529, + "step": 440, + "tokens/total": 57516032, + "tokens/train_per_sec_per_gpu": 17930.45, + "tokens/trainable": 55645416 + }, + { + "epoch": 2.2734177215189875, + "grad_norm": 1.1953125, + "learning_rate": 2.871119526777315e-06, + "loss": 1.028823184967041, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.79777, + "step": 450, + "tokens/total": 58826752, + "tokens/train_per_sec_per_gpu": 17944.17, + "tokens/trainable": 56914256 + }, + { + "epoch": 2.2734177215189875, + "eval_indic_sft_mini_val_loss": 1.0182074308395386, + "eval_indic_sft_mini_val_runtime": 157.9039, + "eval_indic_sft_mini_val_samples_per_second": 12.159, + "eval_indic_sft_mini_val_steps_per_second": 3.04, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 450 + }, + { + "epoch": 2.2734177215189875, + "eval_tulu_sft_mini_val_loss": 2.297419309616089, + "eval_tulu_sft_mini_val_runtime": 92.4484, + "eval_tulu_sft_mini_val_samples_per_second": 12.461, + "eval_tulu_sft_mini_val_steps_per_second": 3.115, + "memory/device_reserved (GiB)": 26.59, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 450 + }, + { + "epoch": 2.3240506329113924, + "grad_norm": 1.21875, + "learning_rate": 2.4981654557803026e-06, + "loss": 1.0112363815307617, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.749, + "step": 460, + "tokens/total": 60137472, + "tokens/train_per_sec_per_gpu": 3954.28, + "tokens/trainable": 58182604 + }, + { + "epoch": 2.3746835443037977, + "grad_norm": 1.265625, + "learning_rate": 2.1476778644440553e-06, + "loss": 0.9997093200683593, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.71749, + "step": 470, + "tokens/total": 61448192, + "tokens/train_per_sec_per_gpu": 17956.34, + "tokens/trainable": 59450748 + }, + { + "epoch": 2.4253164556962026, + "grad_norm": 1.21875, + "learning_rate": 1.820706392332824e-06, + "loss": 1.0180435180664062, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.76777, + "step": 480, + "tokens/total": 62758912, + "tokens/train_per_sec_per_gpu": 17939.84, + "tokens/trainable": 60720528 + }, + { + "epoch": 2.4759493670886075, + "grad_norm": 1.234375, + "learning_rate": 1.518230252982248e-06, + "loss": 1.040645217895508, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.83104, + "step": 490, + "tokens/total": 64069632, + "tokens/train_per_sec_per_gpu": 17931.8, + "tokens/trainable": 61988728 + }, + { + "epoch": 2.526582278481013, + "grad_norm": 1.171875, + "learning_rate": 1.2411553013525457e-06, + "loss": 1.0239150047302246, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.78407, + "step": 500, + "tokens/total": 65380352, + "tokens/train_per_sec_per_gpu": 17916.35, + "tokens/trainable": 63254896 + }, + { + "epoch": 2.526582278481013, + "eval_indic_sft_mini_val_loss": 1.018338680267334, + "eval_indic_sft_mini_val_runtime": 157.6995, + "eval_indic_sft_mini_val_samples_per_second": 12.175, + "eval_indic_sft_mini_val_steps_per_second": 3.044, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 500 + }, + { + "epoch": 2.526582278481013, + "eval_tulu_sft_mini_val_loss": 2.299147605895996, + "eval_tulu_sft_mini_val_runtime": 92.1488, + "eval_tulu_sft_mini_val_samples_per_second": 12.502, + "eval_tulu_sft_mini_val_steps_per_second": 3.125, + "memory/device_reserved (GiB)": 26.59, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 500 + } + ], + "logging_steps": 10, + "max_steps": 591, + "num_input_tokens_seen": 0, + "num_train_epochs": 3, + "save_steps": 500, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 1.4039702725617254e+17, + "train_batch_size": 4, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-500/training_args.bin b/checkpoint-500/training_args.bin new file mode 100644 index 0000000..c9f1551 --- /dev/null +++ b/checkpoint-500/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f1ab83fb7a53e79e01752f8af988eed016a56cd8159a16af0f41fcfaaa076803 +size 9169 diff --git a/checkpoint-591/chat_template.jinja b/checkpoint-591/chat_template.jinja new file mode 100644 index 0000000..28028c0 --- /dev/null +++ b/checkpoint-591/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-591/config.json b/checkpoint-591/config.json new file mode 100644 index 0000000..cf68265 --- /dev/null +++ b/checkpoint-591/config.json @@ -0,0 +1,58 @@ +{ + "architectures": [ + "Qwen2ForCausalLM" + ], + "attention_dropout": 0.0, + "bos_token_id": null, + "dtype": "bfloat16", + "eos_token_id": 151643, + "hidden_act": "silu", + "hidden_size": 896, + "initializer_range": 0.02, + "intermediate_size": 4864, + "layer_types": [ + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention" + ], + "max_position_embeddings": 32768, + "max_window_layers": 24, + "model_type": "qwen2", + "num_attention_heads": 14, + "num_hidden_layers": 24, + "num_key_value_heads": 2, + "pad_token_id": 151643, + "rms_norm_eps": 1e-06, + "rope_parameters": { + "rope_theta": 1000000.0, + "rope_type": "default" + }, + "sliding_window": null, + "tie_word_embeddings": true, + "transformers_version": "5.14.1", + "use_cache": false, + "use_mrope": false, + "use_sliding_window": false, + "vocab_size": 151936 +} diff --git a/checkpoint-591/generation_config.json b/checkpoint-591/generation_config.json new file mode 100644 index 0000000..c2d1ff4 --- /dev/null +++ b/checkpoint-591/generation_config.json @@ -0,0 +1,9 @@ +{ + "do_sample": true, + "eos_token_id": [ + 151643 + ], + "max_new_tokens": 2048, + "pad_token_id": 151643, + "transformers_version": "5.14.1" +} diff --git a/checkpoint-591/model.safetensors b/checkpoint-591/model.safetensors new file mode 100644 index 0000000..f1905fc --- /dev/null +++ b/checkpoint-591/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4f2ef297270ca28cd4a413f624c81a6c0e639cc6667fbbfd6fdab6640a114eb6 +size 988097824 diff --git a/checkpoint-591/tokenizer.json b/checkpoint-591/tokenizer.json new file mode 100644 index 0000000..34510ff --- /dev/null +++ b/checkpoint-591/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3fd169731d2cbde95e10bf356d66d5997fd885dd8dbb6fb4684da3f23b2585d8 +size 11421892 diff --git a/checkpoint-591/tokenizer_config.json b/checkpoint-591/tokenizer_config.json new file mode 100644 index 0000000..df536e9 --- /dev/null +++ b/checkpoint-591/tokenizer_config.json @@ -0,0 +1,30 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "local_files_only": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-591/tokens_state.json b/checkpoint-591/tokens_state.json new file mode 100644 index 0000000..919c752 --- /dev/null +++ b/checkpoint-591/tokens_state.json @@ -0,0 +1 @@ +{"total": 77307904, "trainable": 74795984} \ No newline at end of file diff --git a/checkpoint-591/trainer_state.json b/checkpoint-591/trainer_state.json new file mode 100644 index 0000000..58c76c2 --- /dev/null +++ b/checkpoint-591/trainer_state.json @@ -0,0 +1,1145 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 2.9873417721518987, + "eval_steps": 50, + "global_step": 591, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0, + "eval_indic_sft_mini_val_loss": 1.127184510231018, + "eval_indic_sft_mini_val_runtime": 156.0033, + "eval_indic_sft_mini_val_samples_per_second": 12.307, + "eval_indic_sft_mini_val_steps_per_second": 3.077, + "memory/device_reserved (GiB)": 24.22, + "memory/max_active (GiB)": 24.14, + "memory/max_allocated (GiB)": 24.14, + "step": 0 + }, + { + "epoch": 0, + "eval_tulu_sft_mini_val_loss": 2.330348491668701, + "eval_tulu_sft_mini_val_runtime": 91.8069, + "eval_tulu_sft_mini_val_samples_per_second": 12.548, + "eval_tulu_sft_mini_val_steps_per_second": 3.137, + "memory/device_reserved (GiB)": 24.22, + "memory/max_active (GiB)": 24.14, + "memory/max_allocated (GiB)": 24.14, + "step": 0 + }, + { + "epoch": 0.05063291139240506, + "grad_norm": 1.9375, + "learning_rate": 1.0588235294117648e-05, + "loss": 1.2239381790161132, + "memory/device_reserved (GiB)": 37.02, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 3.40055, + "step": 10, + "tokens/total": 1310720, + "tokens/trainable": 1268231 + }, + { + "epoch": 0.10126582278481013, + "grad_norm": 1.640625, + "learning_rate": 1.9999400896826965e-05, + "loss": 1.1529170989990234, + "memory/device_reserved (GiB)": 37.02, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 3.16742, + "step": 20, + "tokens/total": 2621440, + "tokens/train_per_sec_per_gpu": 17988.68, + "tokens/trainable": 2535045 + }, + { + "epoch": 0.1518987341772152, + "grad_norm": 1.4453125, + "learning_rate": 1.9978439822224228e-05, + "loss": 1.1212153434753418, + "memory/device_reserved (GiB)": 37.02, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 3.06858, + "step": 30, + "tokens/total": 3932160, + "tokens/train_per_sec_per_gpu": 17958.15, + "tokens/trainable": 3804033 + }, + { + "epoch": 0.20253164556962025, + "grad_norm": 1.3828125, + "learning_rate": 1.9927595335238736e-05, + "loss": 1.106326198577881, + "memory/device_reserved (GiB)": 37.02, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 3.02323, + "step": 40, + "tokens/total": 5242880, + "tokens/train_per_sec_per_gpu": 17929.18, + "tokens/trainable": 5070883 + }, + { + "epoch": 0.25316455696202533, + "grad_norm": 1.421875, + "learning_rate": 1.984701970484229e-05, + "loss": 1.0841044425964355, + "memory/device_reserved (GiB)": 37.02, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.95679, + "step": 50, + "tokens/total": 6553600, + "tokens/train_per_sec_per_gpu": 17943.82, + "tokens/trainable": 6341109 + }, + { + "epoch": 0.25316455696202533, + "eval_indic_sft_mini_val_loss": 1.0926785469055176, + "eval_indic_sft_mini_val_runtime": 157.8472, + "eval_indic_sft_mini_val_samples_per_second": 12.164, + "eval_indic_sft_mini_val_steps_per_second": 3.041, + "memory/device_reserved (GiB)": 37.02, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 50 + }, + { + "epoch": 0.25316455696202533, + "eval_tulu_sft_mini_val_loss": 2.3083696365356445, + "eval_tulu_sft_mini_val_runtime": 92.1987, + "eval_tulu_sft_mini_val_samples_per_second": 12.495, + "eval_tulu_sft_mini_val_steps_per_second": 3.124, + "memory/device_reserved (GiB)": 26.58, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 50 + }, + { + "epoch": 0.3037974683544304, + "grad_norm": 1.34375, + "learning_rate": 1.9736954238777793e-05, + "loss": 1.1303999900817872, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 3.09689, + "step": 60, + "tokens/total": 7864320, + "tokens/train_per_sec_per_gpu": 3953.8, + "tokens/trainable": 7607732 + }, + { + "epoch": 0.35443037974683544, + "grad_norm": 1.4296875, + "learning_rate": 1.9597728560891266e-05, + "loss": 1.1227096557617187, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 3.07317, + "step": 70, + "tokens/total": 9175040, + "tokens/train_per_sec_per_gpu": 17981.63, + "tokens/trainable": 8877038 + }, + { + "epoch": 0.4050632911392405, + "grad_norm": 1.3828125, + "learning_rate": 1.9429759623974992e-05, + "loss": 1.109241771697998, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 3.03206, + "step": 80, + "tokens/total": 10485760, + "tokens/train_per_sec_per_gpu": 17934.18, + "tokens/trainable": 10145302 + }, + { + "epoch": 0.45569620253164556, + "grad_norm": 1.2578125, + "learning_rate": 1.9233550461078114e-05, + "loss": 1.071034049987793, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.9184, + "step": 90, + "tokens/total": 11796480, + "tokens/train_per_sec_per_gpu": 17952.48, + "tokens/trainable": 11415878 + }, + { + "epoch": 0.5063291139240507, + "grad_norm": 1.2421875, + "learning_rate": 1.900968867902419e-05, + "loss": 1.0816166877746582, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.94944, + "step": 100, + "tokens/total": 13107200, + "tokens/train_per_sec_per_gpu": 17943.02, + "tokens/trainable": 12684573 + }, + { + "epoch": 0.5063291139240507, + "eval_indic_sft_mini_val_loss": 1.0623797178268433, + "eval_indic_sft_mini_val_runtime": 157.7495, + "eval_indic_sft_mini_val_samples_per_second": 12.171, + "eval_indic_sft_mini_val_steps_per_second": 3.043, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 100 + }, + { + "epoch": 0.5063291139240507, + "eval_tulu_sft_mini_val_loss": 2.2967746257781982, + "eval_tulu_sft_mini_val_runtime": 92.342, + "eval_tulu_sft_mini_val_samples_per_second": 12.475, + "eval_tulu_sft_mini_val_steps_per_second": 3.119, + "memory/device_reserved (GiB)": 26.58, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 100 + }, + { + "epoch": 0.5569620253164557, + "grad_norm": 1.2421875, + "learning_rate": 1.8758844698647457e-05, + "loss": 1.106839370727539, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 3.02478, + "step": 110, + "tokens/total": 14417920, + "tokens/train_per_sec_per_gpu": 3955.64, + "tokens/trainable": 13951995 + }, + { + "epoch": 0.6075949367088608, + "grad_norm": 1.8984375, + "learning_rate": 1.848176974701775e-05, + "loss": 1.0786801338195802, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.9408, + "step": 120, + "tokens/total": 15728640, + "tokens/train_per_sec_per_gpu": 17975.05, + "tokens/trainable": 15221447 + }, + { + "epoch": 0.6582278481012658, + "grad_norm": 1.28125, + "learning_rate": 1.8179293607667177e-05, + "loss": 1.0739567756652832, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.92694, + "step": 130, + "tokens/total": 17039360, + "tokens/train_per_sec_per_gpu": 17954.31, + "tokens/trainable": 16490903 + }, + { + "epoch": 0.7088607594936709, + "grad_norm": 1.2890625, + "learning_rate": 1.7852322135555946e-05, + "loss": 1.0759190559387206, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.93269, + "step": 140, + "tokens/total": 18350080, + "tokens/train_per_sec_per_gpu": 17887.36, + "tokens/trainable": 17757184 + }, + { + "epoch": 0.759493670886076, + "grad_norm": 1.3828125, + "learning_rate": 1.7501834544219697e-05, + "loss": 1.0366504669189454, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.81976, + "step": 150, + "tokens/total": 19660800, + "tokens/train_per_sec_per_gpu": 17911.77, + "tokens/trainable": 19024962 + }, + { + "epoch": 0.759493670886076, + "eval_indic_sft_mini_val_loss": 1.0453351736068726, + "eval_indic_sft_mini_val_runtime": 157.7623, + "eval_indic_sft_mini_val_samples_per_second": 12.17, + "eval_indic_sft_mini_val_steps_per_second": 3.043, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 150 + }, + { + "epoch": 0.759493670886076, + "eval_tulu_sft_mini_val_loss": 2.2883615493774414, + "eval_tulu_sft_mini_val_runtime": 92.4616, + "eval_tulu_sft_mini_val_samples_per_second": 12.459, + "eval_tulu_sft_mini_val_steps_per_second": 3.115, + "memory/device_reserved (GiB)": 26.58, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 150 + }, + { + "epoch": 0.810126582278481, + "grad_norm": 1.328125, + "learning_rate": 1.7128880473222688e-05, + "loss": 1.0812637329101562, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.9484, + "step": 160, + "tokens/total": 20971520, + "tokens/train_per_sec_per_gpu": 3951.59, + "tokens/trainable": 20291324 + }, + { + "epoch": 0.8607594936708861, + "grad_norm": 1.328125, + "learning_rate": 1.6734576844699234e-05, + "loss": 1.0667606353759767, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.90595, + "step": 170, + "tokens/total": 22282240, + "tokens/train_per_sec_per_gpu": 17990.97, + "tokens/trainable": 21559984 + }, + { + "epoch": 0.9113924050632911, + "grad_norm": 1.3515625, + "learning_rate": 1.6320104518397473e-05, + "loss": 1.067934513092041, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.90936, + "step": 180, + "tokens/total": 23592960, + "tokens/train_per_sec_per_gpu": 17944.73, + "tokens/trainable": 22828372 + }, + { + "epoch": 0.9620253164556962, + "grad_norm": 1.296875, + "learning_rate": 1.588670475524283e-05, + "loss": 1.04850492477417, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.85338, + "step": 190, + "tokens/total": 24903680, + "tokens/train_per_sec_per_gpu": 17890.69, + "tokens/trainable": 24094636 + }, + { + "epoch": 1.010126582278481, + "grad_norm": 1.3046875, + "learning_rate": 1.5435675500012212e-05, + "loss": 1.0491924285888672, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.85534, + "step": 200, + "tokens/total": 26136576, + "tokens/train_per_sec_per_gpu": 17465.18, + "tokens/trainable": 25286404 + }, + { + "epoch": 1.010126582278481, + "eval_indic_sft_mini_val_loss": 1.0354256629943848, + "eval_indic_sft_mini_val_runtime": 157.741, + "eval_indic_sft_mini_val_samples_per_second": 12.172, + "eval_indic_sft_mini_val_steps_per_second": 3.043, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 200 + }, + { + "epoch": 1.010126582278481, + "eval_tulu_sft_mini_val_loss": 2.291062593460083, + "eval_tulu_sft_mini_val_runtime": 92.344, + "eval_tulu_sft_mini_val_samples_per_second": 12.475, + "eval_tulu_sft_mini_val_steps_per_second": 3.119, + "memory/device_reserved (GiB)": 26.58, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 200 + }, + { + "epoch": 1.0607594936708862, + "grad_norm": 1.375, + "learning_rate": 1.4968367494251486e-05, + "loss": 1.0487144470214844, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.85398, + "step": 210, + "tokens/total": 27447296, + "tokens/train_per_sec_per_gpu": 3957.26, + "tokens/trainable": 26554084 + }, + { + "epoch": 1.111392405063291, + "grad_norm": 1.3671875, + "learning_rate": 1.4486180231077278e-05, + "loss": 1.0355692863464356, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.81671, + "step": 220, + "tokens/total": 28758016, + "tokens/train_per_sec_per_gpu": 17979.8, + "tokens/trainable": 27823280 + }, + { + "epoch": 1.1620253164556962, + "grad_norm": 1.265625, + "learning_rate": 1.3990557763977694e-05, + "loss": 1.0380287170410156, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.82365, + "step": 230, + "tokens/total": 30068736, + "tokens/train_per_sec_per_gpu": 17964.75, + "tokens/trainable": 29092706 + }, + { + "epoch": 1.2126582278481013, + "grad_norm": 1.2734375, + "learning_rate": 1.3482984382163713e-05, + "loss": 1.036821174621582, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.82024, + "step": 240, + "tokens/total": 31379456, + "tokens/train_per_sec_per_gpu": 17930.26, + "tokens/trainable": 30360732 + }, + { + "epoch": 1.2632911392405064, + "grad_norm": 1.2265625, + "learning_rate": 1.2964980165422701e-05, + "loss": 0.995778751373291, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.70683, + "step": 250, + "tokens/total": 32690176, + "tokens/train_per_sec_per_gpu": 17927.94, + "tokens/trainable": 31631352 + }, + { + "epoch": 1.2632911392405064, + "eval_indic_sft_mini_val_loss": 1.027944803237915, + "eval_indic_sft_mini_val_runtime": 157.8775, + "eval_indic_sft_mini_val_samples_per_second": 12.161, + "eval_indic_sft_mini_val_steps_per_second": 3.04, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 250 + }, + { + "epoch": 1.2632911392405064, + "eval_tulu_sft_mini_val_loss": 2.2984113693237305, + "eval_tulu_sft_mini_val_runtime": 92.3048, + "eval_tulu_sft_mini_val_samples_per_second": 12.48, + "eval_tulu_sft_mini_val_steps_per_second": 3.12, + "memory/device_reserved (GiB)": 26.58, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 250 + }, + { + "epoch": 1.3139240506329113, + "grad_norm": 1.2109375, + "learning_rate": 1.2438096431786408e-05, + "loss": 1.0438777923583984, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.84021, + "step": 260, + "tokens/total": 34000896, + "tokens/train_per_sec_per_gpu": 3955.89, + "tokens/trainable": 32899468 + }, + { + "epoch": 1.3645569620253164, + "grad_norm": 1.28125, + "learning_rate": 1.1903911091646684e-05, + "loss": 1.0240283012390137, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.78439, + "step": 270, + "tokens/total": 35311616, + "tokens/train_per_sec_per_gpu": 17961.08, + "tokens/trainable": 34168384 + }, + { + "epoch": 1.4151898734177215, + "grad_norm": 1.234375, + "learning_rate": 1.1364023922232503e-05, + "loss": 1.031261920928955, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.8046, + "step": 280, + "tokens/total": 36622336, + "tokens/train_per_sec_per_gpu": 17939.56, + "tokens/trainable": 35436704 + }, + { + "epoch": 1.4658227848101266, + "grad_norm": 1.3359375, + "learning_rate": 1.0820051776600175e-05, + "loss": 1.0369884490966796, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.82071, + "step": 290, + "tokens/total": 37933056, + "tokens/train_per_sec_per_gpu": 17936.98, + "tokens/trainable": 36705368 + }, + { + "epoch": 1.5164556962025317, + "grad_norm": 1.1875, + "learning_rate": 1.0273623741484924e-05, + "loss": 1.0238310813903808, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.78384, + "step": 300, + "tokens/total": 39243776, + "tokens/train_per_sec_per_gpu": 17926.64, + "tokens/trainable": 37973400 + }, + { + "epoch": 1.5164556962025317, + "eval_indic_sft_mini_val_loss": 1.0224663019180298, + "eval_indic_sft_mini_val_runtime": 157.771, + "eval_indic_sft_mini_val_samples_per_second": 12.17, + "eval_indic_sft_mini_val_steps_per_second": 3.042, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 300 + }, + { + "epoch": 1.5164556962025317, + "eval_tulu_sft_mini_val_loss": 2.295070171356201, + "eval_tulu_sft_mini_val_runtime": 92.349, + "eval_tulu_sft_mini_val_samples_per_second": 12.474, + "eval_tulu_sft_mini_val_steps_per_second": 3.119, + "memory/device_reserved (GiB)": 26.58, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 300 + }, + { + "epoch": 1.5670886075949366, + "grad_norm": 1.2421875, + "learning_rate": 9.726376258515077e-06, + "loss": 1.0644823074340821, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.89934, + "step": 310, + "tokens/total": 40554496, + "tokens/train_per_sec_per_gpu": 3956.31, + "tokens/trainable": 39240980 + }, + { + "epoch": 1.6177215189873417, + "grad_norm": 1.234375, + "learning_rate": 9.179948223399828e-06, + "loss": 1.0305088996887206, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.80249, + "step": 320, + "tokens/total": 41865216, + "tokens/train_per_sec_per_gpu": 17965.06, + "tokens/trainable": 40509632 + }, + { + "epoch": 1.6683544303797468, + "grad_norm": 1.265625, + "learning_rate": 8.6359760777675e-06, + "loss": 1.0250298500061035, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.78718, + "step": 330, + "tokens/total": 43175936, + "tokens/train_per_sec_per_gpu": 17953.88, + "tokens/trainable": 41777768 + }, + { + "epoch": 1.7189873417721517, + "grad_norm": 1.2265625, + "learning_rate": 8.096088908353316e-06, + "loss": 1.0106207847595214, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.74731, + "step": 340, + "tokens/total": 44486656, + "tokens/train_per_sec_per_gpu": 17933.81, + "tokens/trainable": 43045456 + }, + { + "epoch": 1.769620253164557, + "grad_norm": 1.2578125, + "learning_rate": 7.561903568213595e-06, + "loss": 0.9956131935119629, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.70638, + "step": 350, + "tokens/total": 45797376, + "tokens/train_per_sec_per_gpu": 17927.21, + "tokens/trainable": 44311528 + }, + { + "epoch": 1.769620253164557, + "eval_indic_sft_mini_val_loss": 1.0202709436416626, + "eval_indic_sft_mini_val_runtime": 157.7597, + "eval_indic_sft_mini_val_samples_per_second": 12.17, + "eval_indic_sft_mini_val_steps_per_second": 3.043, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 350 + }, + { + "epoch": 1.769620253164557, + "eval_tulu_sft_mini_val_loss": 2.2989728450775146, + "eval_tulu_sft_mini_val_runtime": 92.4784, + "eval_tulu_sft_mini_val_samples_per_second": 12.457, + "eval_tulu_sft_mini_val_steps_per_second": 3.114, + "memory/device_reserved (GiB)": 26.58, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 350 + }, + { + "epoch": 1.820253164556962, + "grad_norm": 1.2734375, + "learning_rate": 7.035019834577301e-06, + "loss": 1.0437658309936524, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.83989, + "step": 360, + "tokens/total": 47108096, + "tokens/train_per_sec_per_gpu": 3952.58, + "tokens/trainable": 45578288 + }, + { + "epoch": 1.870886075949367, + "grad_norm": 1.2265625, + "learning_rate": 6.517015617836292e-06, + "loss": 1.0040513038635255, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.72932, + "step": 370, + "tokens/total": 48418816, + "tokens/train_per_sec_per_gpu": 17939.3, + "tokens/trainable": 46846324 + }, + { + "epoch": 1.9215189873417722, + "grad_norm": 1.234375, + "learning_rate": 6.009442236022307e-06, + "loss": 1.0151902198791505, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.75989, + "step": 380, + "tokens/total": 49729536, + "tokens/train_per_sec_per_gpu": 17915.04, + "tokens/trainable": 48113428 + }, + { + "epoch": 1.972151898734177, + "grad_norm": 1.3125, + "learning_rate": 5.513819768922723e-06, + "loss": 0.997506046295166, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.71151, + "step": 390, + "tokens/total": 51040256, + "tokens/train_per_sec_per_gpu": 17963.73, + "tokens/trainable": 49382776 + }, + { + "epoch": 2.020253164556962, + "grad_norm": 1.2421875, + "learning_rate": 5.031632505748516e-06, + "loss": 1.0011167526245117, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.72132, + "step": 400, + "tokens/total": 52273152, + "tokens/train_per_sec_per_gpu": 17458.66, + "tokens/trainable": 50572704 + }, + { + "epoch": 2.020253164556962, + "eval_indic_sft_mini_val_loss": 1.0187087059020996, + "eval_indic_sft_mini_val_runtime": 158.0242, + "eval_indic_sft_mini_val_samples_per_second": 12.15, + "eval_indic_sft_mini_val_steps_per_second": 3.038, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 400 + }, + { + "epoch": 2.020253164556962, + "eval_tulu_sft_mini_val_loss": 2.2966930866241455, + "eval_tulu_sft_mini_val_runtime": 92.3222, + "eval_tulu_sft_mini_val_samples_per_second": 12.478, + "eval_tulu_sft_mini_val_steps_per_second": 3.12, + "memory/device_reserved (GiB)": 26.58, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 400 + }, + { + "epoch": 2.070886075949367, + "grad_norm": 1.2421875, + "learning_rate": 4.56432449998779e-06, + "loss": 1.0103323936462403, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.74651, + "step": 410, + "tokens/total": 53583872, + "tokens/train_per_sec_per_gpu": 3952.63, + "tokens/trainable": 51839796 + }, + { + "epoch": 2.1215189873417724, + "grad_norm": 1.234375, + "learning_rate": 4.113295244757171e-06, + "loss": 1.0133058547973632, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.75469, + "step": 420, + "tokens/total": 54894592, + "tokens/train_per_sec_per_gpu": 17984.79, + "tokens/trainable": 53108816 + }, + { + "epoch": 2.1721518987341772, + "grad_norm": 1.21875, + "learning_rate": 3.679895481602529e-06, + "loss": 1.0214984893798829, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.77735, + "step": 430, + "tokens/total": 56205312, + "tokens/train_per_sec_per_gpu": 17959.57, + "tokens/trainable": 54378168 + }, + { + "epoch": 2.222784810126582, + "grad_norm": 1.2109375, + "learning_rate": 3.2654231553007665e-06, + "loss": 0.9840593338012695, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.67529, + "step": 440, + "tokens/total": 57516032, + "tokens/train_per_sec_per_gpu": 17930.45, + "tokens/trainable": 55645416 + }, + { + "epoch": 2.2734177215189875, + "grad_norm": 1.1953125, + "learning_rate": 2.871119526777315e-06, + "loss": 1.028823184967041, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.79777, + "step": 450, + "tokens/total": 58826752, + "tokens/train_per_sec_per_gpu": 17944.17, + "tokens/trainable": 56914256 + }, + { + "epoch": 2.2734177215189875, + "eval_indic_sft_mini_val_loss": 1.0182074308395386, + "eval_indic_sft_mini_val_runtime": 157.9039, + "eval_indic_sft_mini_val_samples_per_second": 12.159, + "eval_indic_sft_mini_val_steps_per_second": 3.04, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 450 + }, + { + "epoch": 2.2734177215189875, + "eval_tulu_sft_mini_val_loss": 2.297419309616089, + "eval_tulu_sft_mini_val_runtime": 92.4484, + "eval_tulu_sft_mini_val_samples_per_second": 12.461, + "eval_tulu_sft_mini_val_steps_per_second": 3.115, + "memory/device_reserved (GiB)": 26.59, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 450 + }, + { + "epoch": 2.3240506329113924, + "grad_norm": 1.21875, + "learning_rate": 2.4981654557803026e-06, + "loss": 1.0112363815307617, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.749, + "step": 460, + "tokens/total": 60137472, + "tokens/train_per_sec_per_gpu": 3954.28, + "tokens/trainable": 58182604 + }, + { + "epoch": 2.3746835443037977, + "grad_norm": 1.265625, + "learning_rate": 2.1476778644440553e-06, + "loss": 0.9997093200683593, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.71749, + "step": 470, + "tokens/total": 61448192, + "tokens/train_per_sec_per_gpu": 17956.34, + "tokens/trainable": 59450748 + }, + { + "epoch": 2.4253164556962026, + "grad_norm": 1.21875, + "learning_rate": 1.820706392332824e-06, + "loss": 1.0180435180664062, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.76777, + "step": 480, + "tokens/total": 62758912, + "tokens/train_per_sec_per_gpu": 17939.84, + "tokens/trainable": 60720528 + }, + { + "epoch": 2.4759493670886075, + "grad_norm": 1.234375, + "learning_rate": 1.518230252982248e-06, + "loss": 1.040645217895508, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.83104, + "step": 490, + "tokens/total": 64069632, + "tokens/train_per_sec_per_gpu": 17931.8, + "tokens/trainable": 61988728 + }, + { + "epoch": 2.526582278481013, + "grad_norm": 1.171875, + "learning_rate": 1.2411553013525457e-06, + "loss": 1.0239150047302246, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.78407, + "step": 500, + "tokens/total": 65380352, + "tokens/train_per_sec_per_gpu": 17916.35, + "tokens/trainable": 63254896 + }, + { + "epoch": 2.526582278481013, + "eval_indic_sft_mini_val_loss": 1.018338680267334, + "eval_indic_sft_mini_val_runtime": 157.6995, + "eval_indic_sft_mini_val_samples_per_second": 12.175, + "eval_indic_sft_mini_val_steps_per_second": 3.044, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 500 + }, + { + "epoch": 2.526582278481013, + "eval_tulu_sft_mini_val_loss": 2.299147605895996, + "eval_tulu_sft_mini_val_runtime": 92.1488, + "eval_tulu_sft_mini_val_samples_per_second": 12.502, + "eval_tulu_sft_mini_val_steps_per_second": 3.125, + "memory/device_reserved (GiB)": 26.59, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 500 + }, + { + "epoch": 2.5772151898734177, + "grad_norm": 1.2109375, + "learning_rate": 9.903113209758098e-07, + "loss": 1.0059453964233398, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.73449, + "step": 510, + "tokens/total": 66691072, + "tokens/train_per_sec_per_gpu": 3921.5, + "tokens/trainable": 64525868 + }, + { + "epoch": 2.6278481012658226, + "grad_norm": 1.25, + "learning_rate": 7.664495389218884e-07, + "loss": 1.0043025016784668, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.73, + "step": 520, + "tokens/total": 68001792, + "tokens/train_per_sec_per_gpu": 17995.95, + "tokens/trainable": 65794468 + }, + { + "epoch": 2.678481012658228, + "grad_norm": 1.2265625, + "learning_rate": 5.702403760250086e-07, + "loss": 0.9985050201416016, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.71422, + "step": 530, + "tokens/total": 69312512, + "tokens/train_per_sec_per_gpu": 17953.05, + "tokens/trainable": 67063008 + }, + { + "epoch": 2.729113924050633, + "grad_norm": 1.234375, + "learning_rate": 4.022714391087379e-07, + "loss": 1.0013395309448243, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.72193, + "step": 540, + "tokens/total": 70623232, + "tokens/train_per_sec_per_gpu": 17930.41, + "tokens/trainable": 68331456 + }, + { + "epoch": 2.779746835443038, + "grad_norm": 1.2109375, + "learning_rate": 2.6304576122221035e-07, + "loss": 1.0258560180664062, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.78948, + "step": 550, + "tokens/total": 71933952, + "tokens/train_per_sec_per_gpu": 17911.01, + "tokens/trainable": 69597800 + }, + { + "epoch": 2.779746835443038, + "eval_indic_sft_mini_val_loss": 1.017600417137146, + "eval_indic_sft_mini_val_runtime": 157.873, + "eval_indic_sft_mini_val_samples_per_second": 12.162, + "eval_indic_sft_mini_val_steps_per_second": 3.04, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 550 + }, + { + "epoch": 2.779746835443038, + "eval_tulu_sft_mini_val_loss": 2.295935869216919, + "eval_tulu_sft_mini_val_runtime": 92.3573, + "eval_tulu_sft_mini_val_samples_per_second": 12.473, + "eval_tulu_sft_mini_val_steps_per_second": 3.118, + "memory/device_reserved (GiB)": 26.59, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 550 + }, + { + "epoch": 2.830379746835443, + "grad_norm": 1.171875, + "learning_rate": 1.5298029515771195e-07, + "loss": 1.001215648651123, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.72159, + "step": 560, + "tokens/total": 73244672, + "tokens/train_per_sec_per_gpu": 3958.43, + "tokens/trainable": 70866632 + }, + { + "epoch": 2.8810126582278484, + "grad_norm": 1.3125, + "learning_rate": 7.24046647612675e-08, + "loss": 0.9905457496643066, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.6927, + "step": 570, + "tokens/total": 74555392, + "tokens/train_per_sec_per_gpu": 17944.54, + "tokens/trainable": 72132904 + }, + { + "epoch": 2.9316455696202532, + "grad_norm": 1.234375, + "learning_rate": 2.156017777577346e-08, + "loss": 1.0318729400634765, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.80632, + "step": 580, + "tokens/total": 75866112, + "tokens/train_per_sec_per_gpu": 17916.79, + "tokens/trainable": 73399168 + }, + { + "epoch": 2.982278481012658, + "grad_norm": 1.2421875, + "learning_rate": 5.991031730367968e-10, + "loss": 1.0094696044921876, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "ppl": 2.74415, + "step": 590, + "tokens/total": 77176832, + "tokens/train_per_sec_per_gpu": 17962.72, + "tokens/trainable": 74669896 + }, + { + "epoch": 2.9873417721518987, + "eval_indic_sft_mini_val_loss": 1.0181607007980347, + "eval_indic_sft_mini_val_runtime": 157.7012, + "eval_indic_sft_mini_val_samples_per_second": 12.175, + "eval_indic_sft_mini_val_steps_per_second": 3.044, + "memory/device_reserved (GiB)": 37.04, + "memory/max_active (GiB)": 32.29, + "memory/max_allocated (GiB)": 32.29, + "step": 591 + }, + { + "epoch": 2.9873417721518987, + "eval_tulu_sft_mini_val_loss": 2.298520088195801, + "eval_tulu_sft_mini_val_runtime": 92.3188, + "eval_tulu_sft_mini_val_samples_per_second": 12.478, + "eval_tulu_sft_mini_val_steps_per_second": 3.12, + "memory/device_reserved (GiB)": 26.59, + "memory/max_active (GiB)": 25.99, + "memory/max_allocated (GiB)": 25.99, + "step": 591 + } + ], + "logging_steps": 10, + "max_steps": 591, + "num_input_tokens_seen": 0, + "num_train_epochs": 3, + "save_steps": 500, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": true + }, + "attributes": {} + } + }, + "total_flos": 1.660101173056635e+17, + "train_batch_size": 4, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-591/training_args.bin b/checkpoint-591/training_args.bin new file mode 100644 index 0000000..c9f1551 --- /dev/null +++ b/checkpoint-591/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f1ab83fb7a53e79e01752f8af988eed016a56cd8159a16af0f41fcfaaa076803 +size 9169 diff --git a/config.json b/config.json new file mode 100644 index 0000000..cf68265 --- /dev/null +++ b/config.json @@ -0,0 +1,58 @@ +{ + "architectures": [ + "Qwen2ForCausalLM" + ], + "attention_dropout": 0.0, + "bos_token_id": null, + "dtype": "bfloat16", + "eos_token_id": 151643, + "hidden_act": "silu", + "hidden_size": 896, + "initializer_range": 0.02, + "intermediate_size": 4864, + "layer_types": [ + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention" + ], + "max_position_embeddings": 32768, + "max_window_layers": 24, + "model_type": "qwen2", + "num_attention_heads": 14, + "num_hidden_layers": 24, + "num_key_value_heads": 2, + "pad_token_id": 151643, + "rms_norm_eps": 1e-06, + "rope_parameters": { + "rope_theta": 1000000.0, + "rope_type": "default" + }, + "sliding_window": null, + "tie_word_embeddings": true, + "transformers_version": "5.14.1", + "use_cache": false, + "use_mrope": false, + "use_sliding_window": false, + "vocab_size": 151936 +} diff --git a/debug.log b/debug.log new file mode 100644 index 0000000..cd81ff6 --- /dev/null +++ b/debug.log @@ -0,0 +1,10560 @@ +[2026-07-30 07:55:41,881] [DEBUG] [axolotl.utils.config.resolve_dtype:168] [PID:4409] bf16 support detected, enabling for this configuration. +[2026-07-30 07:55:46,939] [DEBUG] [axolotl.utils.config.log_gpu_memory_usage:127] [PID:4409] baseline 0.000GB () +[2026-07-30 07:55:46,940] [INFO] [axolotl.cli.config.load_cfg:336] [PID:4409] config: +{ + "activation_offloading": false, + "attn_decontaminates_packing": true, + "attn_implementation": "flash_attention_2", + "attn_needs_dtype_cast": true, + "attn_supports_packing": true, + "attn_uses_flash_lib": true, + "axolotl_config_path": "/workspace/axolotl/qwen2.5_0.5b_cpt_full.yml", + "base_model": "adityabanerjee13/qwen2.5-0.5b-cpt-mix-1to2", + "base_model_config": "adityabanerjee13/qwen2.5-0.5b-cpt-mix-1to2", + "batch_size": 32, + "bf16": true, + "capabilities": { + "bf16": true, + "compute_capability": "sm_89", + "fp8": false, + "n_gpu": 1, + "n_node": 1, + "tf32": true + }, + "chat_template": "tokenizer_default", + "context_parallel_size": 1, + "dataloader_num_workers": 1, + "dataloader_pin_memory": true, + "dataloader_prefetch_factor": 256, + "dataset_num_proc": 1, + "dataset_prepared_path": "./last_run_prepared_1to2", + "datasets": [ + { + "chat_template": "tokenizer_default", + "field_messages": "messages", + "message_property_mappings": { + "content": "content", + "role": "role" + }, + "path": "adityabanerjee13/indic-sft-mini-train", + "split": "train", + "trust_remote_code": false, + "type": "chat_template" + }, + { + "chat_template": "tokenizer_default", + "field_messages": "messages", + "message_property_mappings": { + "content": "content", + "role": "role" + }, + "path": "adityabanerjee13/tulu-sft-mini-train", + "split": "train", + "trust_remote_code": false, + "type": "chat_template" + } + ], + "ddp": false, + "device": "cuda:0", + "dion_rank_fraction": 1.0, + "dion_rank_multiple_of": 1, + "eaft_alpha": 1.0, + "eaft_k": 20, + "env_capabilities": { + "torch_version": "2.12.0" + }, + "eval_batch_size": 4, + "eval_causal_lm_metrics": [ + "sacrebleu", + "comet", + "ter", + "chrf" + ], + "eval_max_new_tokens": 128, + "eval_sample_packing": false, + "eval_steps": 0.08333333333333333, + "eval_table_size": 0, + "evals_per_epoch": 4, + "experimental_skip_move_to_device": true, + "fp16": false, + "generate_samples": false, + "generation_do_sample": true, + "generation_max_new_tokens": 50, + "generation_prompt_ratio": 0.5, + "generation_temperature": 0.7, + "gradient_accumulation_steps": 8, + "gradient_checkpointing": true, + "gradient_checkpointing_kwargs": { + "use_reentrant": true + }, + "group_by_length": false, + "hub_model_id": "adityabanerjee13/qwen2.5-0.5b-sft-IT", + "hub_strategy": "all_checkpoints", + "include_tkps": true, + "is_falcon_derived_model": false, + "is_llama_derived_model": false, + "is_mistral_derived_model": false, + "layer_offloading": false, + "learning_rate": 2e-05, + "lisa_layers_attribute": "model.layers", + "load_best_model_at_end": false, + "load_in_4bit": false, + "load_in_8bit": false, + "local_rank": 0, + "logging_steps": 10, + "lora_dropout": 0.0, + "loraplus_lr_embedding": 1e-06, + "lr_scheduler": "cosine", + "max_grad_norm": 1.0, + "mean_resizing_embeddings": false, + "merge_method": "memory_efficient", + "micro_batch_size": 4, + "model_config_type": "qwen2", + "num_epochs": 3.0, + "num_generation_samples": 3, + "optimizer": "adamw_torch_fused", + "otel_metrics_host": "localhost", + "otel_metrics_port": 8000, + "output_dir": "./outputs/qwen2.5-0.5b-sft-IT", + "pad_to_sequence_len": true, + "plugins": [ + "multi_eval_plugin.MultiEvalPlugin" + ], + "pretrain_multipack_attn": true, + "profiler_steps_start": 0, + "qgalore_cos_threshold": 0.4, + "qgalore_gamma_proj": 2, + "qgalore_proj_bits": 4, + "qgalore_proj_group_size": 256, + "qgalore_proj_quant": true, + "qgalore_proj_type": "std", + "qgalore_queue_size": 5, + "qgalore_rank": 256, + "qgalore_scale": 0.25, + "qgalore_update_proj_gap": 200, + "qlora_sharded_model_loading": false, + "quantize_moe_experts": false, + "ray_num_workers": 1, + "relora_prune_method": "magnitude", + "remove_unused_columns": false, + "resources_per_worker": { + "GPU": 1 + }, + "sample_packing": true, + "sample_packing_bin_size": 200, + "sample_packing_group_size": 100000, + "save_only_model": true, + "save_safetensors": true, + "save_steps": 500, + "save_strategy": "steps", + "save_total_limit": 30, + "sequence_len": 4096, + "shuffle_before_merging_datasets": false, + "shuffle_merged_datasets": true, + "skip_prepare_dataset": false, + "streaming_multipack_buffer_size": 10000, + "strict": false, + "tensor_parallel_size": 1, + "test_datasets": [ + { + "chat_template": "tokenizer_default", + "field_messages": "messages", + "message_property_mappings": { + "content": "content", + "role": "role" + }, + "path": "adityabanerjee13/indic-sft-mini-val", + "split": "validation", + "trust_remote_code": false, + "type": "chat_template" + }, + { + "chat_template": "tokenizer_default", + "field_messages": "messages", + "message_property_mappings": { + "content": "content", + "role": "role" + }, + "path": "adityabanerjee13/tulu-sft-mini-val", + "split": "validation", + "trust_remote_code": false, + "type": "chat_template" + } + ], + "tf32": true, + "tiled_mlp_use_original_mlp": true, + "tokenizer_config": "adityabanerjee13/qwen2.5-0.5b-cpt-mix-1to2", + "tokenizer_save_jinja_files": true, + "tokenizer_type": "AutoTokenizer", + "torch_dtype": "torch.bfloat16", + "train_on_inputs": true, + "trl": { + "async_prefetch": false, + "log_completions": false, + "mask_truncated_completions": false, + "ref_model_mixup_alpha": 0.9, + "ref_model_sync_steps": 64, + "replay_buffer_size": 0, + "replay_recompute_logps": true, + "reroll_max_groups": 1, + "reroll_start_fraction": 1.0, + "reward_num_workers": 1, + "scale_rewards": true, + "skip_zero_advantage_batches": true, + "sync_ref_model": false, + "use_data_producer": false, + "use_vllm": false, + "vllm_lora_sync": false, + "vllm_server_host": "0.0.0.0", + "vllm_server_port": 8000 + }, + "trust_remote_code": false, + "type_of_model": "AutoModelForCausalLM", + "use_otel_metrics": false, + "use_ray": false, + "use_wandb": true, + "val_set_size": 0.0, + "vllm": { + "device": "auto", + "dtype": "auto", + "gpu_memory_utilization": 0.9, + "host": "0.0.0.0", + "port": 8000 + }, + "wandb_entity": "models-na9841", + "wandb_log_model": "false", + "wandb_name": "qwen2.5-0.5b-sft-IT", + "wandb_project": "indic-sft", + "warmup_ratio": 0.03, + "weight_decay": 0.01, + "world_size": 1 +} +[2026-07-30 07:55:49,832] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:311] [PID:4409] EOS: 151643 / <|endoftext|> +[2026-07-30 07:55:49,832] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:312] [PID:4409] BOS: None / None +[2026-07-30 07:55:49,832] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:313] [PID:4409] PAD: 151643 / <|endoftext|> +[2026-07-30 07:55:49,832] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:314] [PID:4409] UNK: None / None +[2026-07-30 07:55:49,833] [INFO] [axolotl.utils.data.shared.load_preprocessed_dataset:538] [PID:4409] Loading prepared dataset from disk at last_run_prepared_1to2/5a81e764811e8dde8c7e6e3ccbf1cc5b... +[2026-07-30 07:55:49,838] [INFO] [axolotl.utils.data.shared.load_preprocessed_dataset:538] [PID:4409] Loading prepared dataset from disk at last_run_prepared_1to2/71c923a31f92df827fe79509fe698145... +[2026-07-30 07:55:49,841] [DEBUG] [axolotl.utils.trainer.calculate_total_num_steps:418] [PID:4409] total_num_tokens: 25_845_531 +[2026-07-30 07:55:49,885] [DEBUG] [axolotl.utils.trainer.calculate_total_num_steps:440] [PID:4409] total_supervised_tokens: 25_032_901 +[2026-07-30 07:55:50,118] [DEBUG] [axolotl.utils.samplers.multipack.pack_parallel:177] [PID:4409] Using single process for pack_parallel, running sequentially. +[2026-07-30 07:55:50,678] [DEBUG] [axolotl.utils.samplers.multipack.pack_parallel:177] [PID:4409] Using single process for pack_parallel, running sequentially. +[2026-07-30 07:55:50,882] [DEBUG] [axolotl.utils.samplers.multipack.__len__:462] [PID:4409] generate_batches time: 0.20723891258239746 +[2026-07-30 07:55:50,886] [DEBUG] [axolotl.utils.samplers.multipack.pack_parallel:177] [PID:4409] Using single process for pack_parallel, running sequentially. +[2026-07-30 07:55:51,097] [DEBUG] [axolotl.utils.samplers.multipack.__len__:462] [PID:4409] generate_batches time: 0.21416831016540527 +[2026-07-30 07:55:51,100] [DEBUG] [axolotl.utils.samplers.multipack.pack_parallel:177] [PID:4409] Using single process for pack_parallel, running sequentially. +[2026-07-30 07:55:51,322] [DEBUG] [axolotl.utils.samplers.multipack.__len__:462] [PID:4409] generate_batches time: 0.22472453117370605 +[2026-07-30 07:55:51,326] [DEBUG] [axolotl.utils.samplers.multipack.pack_parallel:177] [PID:4409] Using single process for pack_parallel, running sequentially. +[2026-07-30 07:55:51,540] [DEBUG] [axolotl.utils.samplers.multipack.__len__:462] [PID:4409] generate_batches time: 0.21713471412658691 +[2026-07-30 07:55:51,563] [INFO] [axolotl.utils.samplers.multipack.calc_min_len:438] [PID:4409] gather_len_batches: [1580] +[2026-07-30 07:55:51,563] [DEBUG] [axolotl.utils.trainer.calculate_total_num_steps:497] [PID:4409] data_loader_len: 197 +[2026-07-30 07:55:51,563] [INFO] [axolotl.utils.trainer.calc_sample_packing_eff_est:506] [PID:4409] sample_packing_eff_est across ranks: [0.9984088752843157] +[2026-07-30 07:55:51,563] [DEBUG] [axolotl.utils.trainer.calculate_total_num_steps:518] [PID:4409] sample_packing_eff_est: 1.0 +[2026-07-30 07:55:51,563] [DEBUG] [axolotl.utils.trainer.calculate_total_num_steps:523] [PID:4409] total_num_steps: 591 +[2026-07-30 07:55:51,563] [INFO] [axolotl.utils.data.sft._prepare_standard_dataset:121] [PID:4409] Maximum number of steps set at 591 +[2026-07-30 07:55:51,594] [DEBUG] [axolotl.train.setup_model_and_tokenizer:70] [PID:4409] loading tokenizer... adityabanerjee13/qwen2.5-0.5b-cpt-mix-1to2 +[2026-07-30 07:55:52,551] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:311] [PID:4409] EOS: 151643 / <|endoftext|> +[2026-07-30 07:55:52,551] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:312] [PID:4409] BOS: None / None +[2026-07-30 07:55:52,551] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:313] [PID:4409] PAD: 151643 / <|endoftext|> +[2026-07-30 07:55:52,551] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:314] [PID:4409] UNK: None / None +[2026-07-30 07:55:52,551] [DEBUG] [axolotl.train.setup_model_and_tokenizer:81] [PID:4409] Loading model +[2026-07-30 07:55:52,574] [DEBUG] [axolotl.monkeypatch.torchao_optim.patch_torchao_optim_state_8bit:75] [PID:4409] Patched OptimState8bit for torch.compile compatibility +[2026-07-30 07:55:52,574] [DEBUG] [axolotl.monkeypatch.torchao_optim.patch_torchao_optim_state_8bit:122] [PID:4409] Patched OptimState4bit for torch.compile compatibility +[2026-07-30 07:55:52,574] [DEBUG] [axolotl.monkeypatch.torchao_optim.patch_torchao_optim_state_8bit:154] [PID:4409] Patched OptimStateFp8 for torch.compile compatibility +[2026-07-30 07:55:52,578] [DEBUG] [axolotl.monkeypatch.transformers.trainer_loss_calc.patch_evaluation_loop:94] [PID:4409] Patched Trainer.evaluation_loop with nanmean loss calculation +[2026-07-30 07:55:52,579] [DEBUG] [axolotl.monkeypatch.transformers.trainer_loss_calc.patch_maybe_log_save_evaluate:148] [PID:4409] Patched Trainer._maybe_log_save_evaluate with nanmean loss calculation +[2026-07-30 07:55:52,580] [INFO] [axolotl.loaders.patch_manager._apply_multipack_patches:922] [PID:4409] Applying multipack dataloader patch for sample packing... +[2026-07-30 07:55:52,660] [WARNING] [kernels._versions.resolve_version_spec_as_ref:83] [PID:4409] You are using version 1 of 'kernels-community/flash-attn2', but version 3 is available. + Fetching ... files: 0it [00:00, ?it/s] Fetching ... files: 23it [00:00, 149564.33it/s] + Loading weights: 0%| | 0/290 [00:00", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "local_files_only": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/training_args.bin b/training_args.bin new file mode 100644 index 0000000..c9f1551 --- /dev/null +++ b/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f1ab83fb7a53e79e01752f8af988eed016a56cd8159a16af0f41fcfaaa076803 +size 9169