初始化项目,由ModelHub XC社区提供模型

Model: adityabanerjee13/qwen2.5-0.5b-sft-IT
Source: Original Platform
This commit is contained in:
ModelHub XC
2026-09-13 17:18:17 +08:00
commit ca60016a55
28 changed files with 13377 additions and 0 deletions

36
.gitattributes vendored Normal file
View File

@@ -0,0 +1,36 @@
*.7z filter=lfs diff=lfs merge=lfs -text
*.arrow filter=lfs diff=lfs merge=lfs -text
*.bin filter=lfs diff=lfs merge=lfs -text
*.bz2 filter=lfs diff=lfs merge=lfs -text
*.ckpt filter=lfs diff=lfs merge=lfs -text
*.ftz filter=lfs diff=lfs merge=lfs -text
*.gz filter=lfs diff=lfs merge=lfs -text
*.h5 filter=lfs diff=lfs merge=lfs -text
*.joblib filter=lfs diff=lfs merge=lfs -text
*.lfs.* filter=lfs diff=lfs merge=lfs -text
*.mlmodel filter=lfs diff=lfs merge=lfs -text
*.model filter=lfs diff=lfs merge=lfs -text
*.msgpack filter=lfs diff=lfs merge=lfs -text
*.npy filter=lfs diff=lfs merge=lfs -text
*.npz filter=lfs diff=lfs merge=lfs -text
*.onnx filter=lfs diff=lfs merge=lfs -text
*.ot filter=lfs diff=lfs merge=lfs -text
*.parquet filter=lfs diff=lfs merge=lfs -text
*.pb filter=lfs diff=lfs merge=lfs -text
*.pickle filter=lfs diff=lfs merge=lfs -text
*.pkl filter=lfs diff=lfs merge=lfs -text
*.pt filter=lfs diff=lfs merge=lfs -text
*.pth filter=lfs diff=lfs merge=lfs -text
*.rar filter=lfs diff=lfs merge=lfs -text
*.safetensors filter=lfs diff=lfs merge=lfs -text
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
*.tar.* filter=lfs diff=lfs merge=lfs -text
*.tar filter=lfs diff=lfs merge=lfs -text
*.tflite filter=lfs diff=lfs merge=lfs -text
*.tgz filter=lfs diff=lfs merge=lfs -text
*.wasm filter=lfs diff=lfs merge=lfs -text
*.xz filter=lfs diff=lfs merge=lfs -text
*.zip filter=lfs diff=lfs merge=lfs -text
*.zst filter=lfs diff=lfs merge=lfs -text
*tfevents* filter=lfs diff=lfs merge=lfs -text
tokenizer.json filter=lfs diff=lfs merge=lfs -text

179
README.md Normal file
View File

@@ -0,0 +1,179 @@
---
library_name: transformers
license: apache-2.0
base_model: adityabanerjee13/qwen2.5-0.5b-cpt-mix-1to2
tags:
- axolotl
- generated_from_trainer
datasets:
- adityabanerjee13/indic-sft-mini-train
- adityabanerjee13/tulu-sft-mini-train
model-index:
- name: qwen2.5-0.5b-sft-IT
results: []
---
<!-- This model card has been generated automatically according to the information the Trainer had access to. You
should probably proofread and complete it, then remove this comment. -->
[<img src="https://raw.githubusercontent.com/axolotl-ai-cloud/axolotl/main/image/axolotl-badge-web.png" alt="Built with Axolotl" width="200" height="32"/>](https://github.com/axolotl-ai-cloud/axolotl)
<details><summary>See axolotl config</summary>
axolotl version: `0.19.0.dev0`
```yaml
# ==============================================================================
# Axolotl CPT config — Qwen2.5-0.5B, full-parameter, single GPU.
# Data mix RATIO EXPERIMENT (character-level exact):
#
# RUN 2 of 3 — fineweb : indic = 1 : 2 (FineWeb is HALF the Indic size)
# FineWeb web-crawl chars == Indic train chars / 2.
#
# Indic train : adityabanerjee13/indic-cpt-mini-train (7,907,882 chars)
# FineWeb train: adityabanerjee13/fineweb-cpt-half (3,953,941 chars)
# Validation : adityabanerjee13/indic-cpt-mini-val (held-out 1% Indic)
#
# The datasets are pre-sized to exact character counts on the Hub, so loading
# each one whole gives the exact 1:2 ratio — no slicing needed.
#
# Usage:
# python train.py --config qwen2.5_0.5b_cpt_mix_1to2.yml
# ==============================================================================
base_model: adityabanerjee13/qwen2.5-0.5b-cpt-mix-1to2
model_type: AutoModelForCausalLM
tokenizer_type: AutoTokenizer
trust_remote_code: false
adapter:
load_in_8bit: false
load_in_4bit: false
# multi_eval_plugin splits test_datasets back into per-source eval sets so
# this run logs eval_indic_cpt_mini_val_loss and eval_fineweb_cpt_val_loss
# separately (instead of one merged eval_loss) at every eval step, incl. to
# wandb. Requires this folder on PYTHONPATH — launch via
# `python train.py --config <this file>`.
plugins:
- multi_eval_plugin.MultiEvalPlugin
datasets:
- path: adityabanerjee13/indic-sft-mini-train
type: chat_template
field_messages: messages
split: train
- path: adityabanerjee13/tulu-sft-mini-train
type: chat_template
field_messages: messages
split: train
test_datasets:
- path: adityabanerjee13/indic-sft-mini-val
type: chat_template
field_messages: messages
split: validation
- path: adityabanerjee13/tulu-sft-mini-val
type: chat_template
field_messages: messages
split: validation
train_on_inputs: false
chat_template: tokenizer_default
dataset_prepared_path: ./last_run_prepared_1to2
dataset_num_proc: 1 # single-process tokenize: avoids fork deadlock
val_set_size: 0
output_dir: ./outputs/qwen2.5-0.5b-sft-IT
# --- Sequence packing -----------------------------------------------------
sequence_len: 4096
sample_packing: true
pad_to_sequence_len: true
eval_sample_packing: false
# --- Optimization ----------------------------------------------------------
gradient_accumulation_steps: 8
micro_batch_size: 4
num_epochs: 3
optimizer: adamw_torch_fused
lr_scheduler: cosine
learning_rate: 2e-5
warmup_ratio: 0.03
weight_decay: 0.01
max_grad_norm: 1.0
train_on_inputs: true
group_by_length: false
# --- Precision / memory ---------------------------------------------------
bf16: auto
fp16:
tf32: true
gradient_checkpointing: true
flash_attention: true
# --- Logging / checkpoints ------------------------------------------------
logging_steps: 10
save_strategy: steps
save_steps: 500
save_total_limit: 30
save_only_model: true # save weights only — no optimizer/scheduler state
# (checkpoints ~1/3 the size; can't resume training)
evals_per_epoch: 4
wandb_project: indic-sft
wandb_entity: models-na9841
wandb_name: qwen2.5-0.5b-sft-IT
wandb_log_model: "false"
hub_model_id: adityabanerjee13/qwen2.5-0.5b-sft-IT
hub_strategy: all_checkpoints
special_tokens:
```
</details><br>
# qwen2.5-0.5b-sft-IT
This model is a fine-tuned version of [adityabanerjee13/qwen2.5-0.5b-cpt-mix-1to2](https://huggingface.co/adityabanerjee13/qwen2.5-0.5b-cpt-mix-1to2) on the adityabanerjee13/indic-sft-mini-train and the adityabanerjee13/tulu-sft-mini-train datasets.
## Model description
More information needed
## Intended uses & limitations
More information needed
## Training and evaluation data
More information needed
## Training procedure
### Training hyperparameters
The following hyperparameters were used during training:
- learning_rate: 2e-05
- train_batch_size: 4
- eval_batch_size: 4
- seed: 42
- gradient_accumulation_steps: 8
- total_train_batch_size: 32
- optimizer: Use OptimizerNames.ADAMW_TORCH_FUSED with betas=(0.9,0.999) and epsilon=1e-08 and optimizer_args=No additional optimizer arguments
- lr_scheduler_type: cosine
- lr_scheduler_warmup_steps: 17
- training_steps: 591
### Training results
### Framework versions
- Transformers 5.14.1
- Pytorch 2.12.0+cu130
- Datasets 4.8.4
- Tokenizers 0.22.2

54
chat_template.jinja Normal file
View File

@@ -0,0 +1,54 @@
{%- if tools %}
{{- '<|im_start|>system\n' }}
{%- if messages[0]['role'] == 'system' %}
{{- messages[0]['content'] }}
{%- else %}
{{- 'You are a helpful assistant.' }}
{%- endif %}
{{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
{%- for tool in tools %}
{{- "\n" }}
{{- tool | tojson }}
{%- endfor %}
{{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
{%- else %}
{%- if messages[0]['role'] == 'system' %}
{{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
{%- else %}
{{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
{%- endif %}
{%- endif %}
{%- for message in messages %}
{%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
{{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
{%- elif message.role == "assistant" %}
{{- '<|im_start|>' + message.role }}
{%- if message.content %}
{{- '\n' + message.content }}
{%- endif %}
{%- for tool_call in message.tool_calls %}
{%- if tool_call.function is defined %}
{%- set tool_call = tool_call.function %}
{%- endif %}
{{- '\n<tool_call>\n{"name": "' }}
{{- tool_call.name }}
{{- '", "arguments": ' }}
{{- tool_call.arguments | tojson }}
{{- '}\n</tool_call>' }}
{%- endfor %}
{{- '<|im_end|>\n' }}
{%- elif message.role == "tool" %}
{%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
{{- '<|im_start|>user' }}
{%- endif %}
{{- '\n<tool_response>\n' }}
{{- message.content }}
{{- '\n</tool_response>' }}
{%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
{{- '<|im_end|>\n' }}
{%- endif %}
{%- endif %}
{%- endfor %}
{%- if add_generation_prompt %}
{{- '<|im_start|>assistant\n' }}
{%- endif %}

View File

@@ -0,0 +1,54 @@
{%- if tools %}
{{- '<|im_start|>system\n' }}
{%- if messages[0]['role'] == 'system' %}
{{- messages[0]['content'] }}
{%- else %}
{{- 'You are a helpful assistant.' }}
{%- endif %}
{{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
{%- for tool in tools %}
{{- "\n" }}
{{- tool | tojson }}
{%- endfor %}
{{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
{%- else %}
{%- if messages[0]['role'] == 'system' %}
{{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
{%- else %}
{{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
{%- endif %}
{%- endif %}
{%- for message in messages %}
{%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
{{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
{%- elif message.role == "assistant" %}
{{- '<|im_start|>' + message.role }}
{%- if message.content %}
{{- '\n' + message.content }}
{%- endif %}
{%- for tool_call in message.tool_calls %}
{%- if tool_call.function is defined %}
{%- set tool_call = tool_call.function %}
{%- endif %}
{{- '\n<tool_call>\n{"name": "' }}
{{- tool_call.name }}
{{- '", "arguments": ' }}
{{- tool_call.arguments | tojson }}
{{- '}\n</tool_call>' }}
{%- endfor %}
{{- '<|im_end|>\n' }}
{%- elif message.role == "tool" %}
{%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
{{- '<|im_start|>user' }}
{%- endif %}
{{- '\n<tool_response>\n' }}
{{- message.content }}
{{- '\n</tool_response>' }}
{%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
{{- '<|im_end|>\n' }}
{%- endif %}
{%- endif %}
{%- endfor %}
{%- if add_generation_prompt %}
{{- '<|im_start|>assistant\n' }}
{%- endif %}

View File

@@ -0,0 +1,58 @@
{
"architectures": [
"Qwen2ForCausalLM"
],
"attention_dropout": 0.0,
"bos_token_id": null,
"dtype": "bfloat16",
"eos_token_id": 151643,
"hidden_act": "silu",
"hidden_size": 896,
"initializer_range": 0.02,
"intermediate_size": 4864,
"layer_types": [
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention"
],
"max_position_embeddings": 32768,
"max_window_layers": 24,
"model_type": "qwen2",
"num_attention_heads": 14,
"num_hidden_layers": 24,
"num_key_value_heads": 2,
"pad_token_id": 151643,
"rms_norm_eps": 1e-06,
"rope_parameters": {
"rope_theta": 1000000.0,
"rope_type": "default"
},
"sliding_window": null,
"tie_word_embeddings": true,
"transformers_version": "5.14.1",
"use_cache": false,
"use_mrope": false,
"use_sliding_window": false,
"vocab_size": 151936
}

View File

@@ -0,0 +1,9 @@
{
"do_sample": true,
"eos_token_id": [
151643
],
"max_new_tokens": 2048,
"pad_token_id": 151643,
"transformers_version": "5.14.1"
}

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:9df9ccbbb1556eb3db2d7023d7cef1f34ce7492419e30bad0da7dcb729885eab
size 988097824

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:3fd169731d2cbde95e10bf356d66d5997fd885dd8dbb6fb4684da3f23b2585d8
size 11421892

View File

@@ -0,0 +1,30 @@
{
"add_prefix_space": false,
"backend": "tokenizers",
"bos_token": null,
"clean_up_tokenization_spaces": false,
"eos_token": "<|endoftext|>",
"errors": "replace",
"extra_special_tokens": [
"<|im_start|>",
"<|im_end|>",
"<|object_ref_start|>",
"<|object_ref_end|>",
"<|box_start|>",
"<|box_end|>",
"<|quad_start|>",
"<|quad_end|>",
"<|vision_start|>",
"<|vision_end|>",
"<|vision_pad|>",
"<|image_pad|>",
"<|video_pad|>"
],
"is_local": false,
"local_files_only": false,
"model_max_length": 131072,
"pad_token": "<|endoftext|>",
"split_special_tokens": false,
"tokenizer_class": "Qwen2Tokenizer",
"unk_token": null
}

View File

@@ -0,0 +1 @@
{"total": 65380352, "trainable": 63254896}

View File

@@ -0,0 +1,975 @@
{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 2.526582278481013,
"eval_steps": 50,
"global_step": 500,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0,
"eval_indic_sft_mini_val_loss": 1.127184510231018,
"eval_indic_sft_mini_val_runtime": 156.0033,
"eval_indic_sft_mini_val_samples_per_second": 12.307,
"eval_indic_sft_mini_val_steps_per_second": 3.077,
"memory/device_reserved (GiB)": 24.22,
"memory/max_active (GiB)": 24.14,
"memory/max_allocated (GiB)": 24.14,
"step": 0
},
{
"epoch": 0,
"eval_tulu_sft_mini_val_loss": 2.330348491668701,
"eval_tulu_sft_mini_val_runtime": 91.8069,
"eval_tulu_sft_mini_val_samples_per_second": 12.548,
"eval_tulu_sft_mini_val_steps_per_second": 3.137,
"memory/device_reserved (GiB)": 24.22,
"memory/max_active (GiB)": 24.14,
"memory/max_allocated (GiB)": 24.14,
"step": 0
},
{
"epoch": 0.05063291139240506,
"grad_norm": 1.9375,
"learning_rate": 1.0588235294117648e-05,
"loss": 1.2239381790161132,
"memory/device_reserved (GiB)": 37.02,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 3.40055,
"step": 10,
"tokens/total": 1310720,
"tokens/trainable": 1268231
},
{
"epoch": 0.10126582278481013,
"grad_norm": 1.640625,
"learning_rate": 1.9999400896826965e-05,
"loss": 1.1529170989990234,
"memory/device_reserved (GiB)": 37.02,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 3.16742,
"step": 20,
"tokens/total": 2621440,
"tokens/train_per_sec_per_gpu": 17988.68,
"tokens/trainable": 2535045
},
{
"epoch": 0.1518987341772152,
"grad_norm": 1.4453125,
"learning_rate": 1.9978439822224228e-05,
"loss": 1.1212153434753418,
"memory/device_reserved (GiB)": 37.02,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 3.06858,
"step": 30,
"tokens/total": 3932160,
"tokens/train_per_sec_per_gpu": 17958.15,
"tokens/trainable": 3804033
},
{
"epoch": 0.20253164556962025,
"grad_norm": 1.3828125,
"learning_rate": 1.9927595335238736e-05,
"loss": 1.106326198577881,
"memory/device_reserved (GiB)": 37.02,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 3.02323,
"step": 40,
"tokens/total": 5242880,
"tokens/train_per_sec_per_gpu": 17929.18,
"tokens/trainable": 5070883
},
{
"epoch": 0.25316455696202533,
"grad_norm": 1.421875,
"learning_rate": 1.984701970484229e-05,
"loss": 1.0841044425964355,
"memory/device_reserved (GiB)": 37.02,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.95679,
"step": 50,
"tokens/total": 6553600,
"tokens/train_per_sec_per_gpu": 17943.82,
"tokens/trainable": 6341109
},
{
"epoch": 0.25316455696202533,
"eval_indic_sft_mini_val_loss": 1.0926785469055176,
"eval_indic_sft_mini_val_runtime": 157.8472,
"eval_indic_sft_mini_val_samples_per_second": 12.164,
"eval_indic_sft_mini_val_steps_per_second": 3.041,
"memory/device_reserved (GiB)": 37.02,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 50
},
{
"epoch": 0.25316455696202533,
"eval_tulu_sft_mini_val_loss": 2.3083696365356445,
"eval_tulu_sft_mini_val_runtime": 92.1987,
"eval_tulu_sft_mini_val_samples_per_second": 12.495,
"eval_tulu_sft_mini_val_steps_per_second": 3.124,
"memory/device_reserved (GiB)": 26.58,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 50
},
{
"epoch": 0.3037974683544304,
"grad_norm": 1.34375,
"learning_rate": 1.9736954238777793e-05,
"loss": 1.1303999900817872,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 3.09689,
"step": 60,
"tokens/total": 7864320,
"tokens/train_per_sec_per_gpu": 3953.8,
"tokens/trainable": 7607732
},
{
"epoch": 0.35443037974683544,
"grad_norm": 1.4296875,
"learning_rate": 1.9597728560891266e-05,
"loss": 1.1227096557617187,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 3.07317,
"step": 70,
"tokens/total": 9175040,
"tokens/train_per_sec_per_gpu": 17981.63,
"tokens/trainable": 8877038
},
{
"epoch": 0.4050632911392405,
"grad_norm": 1.3828125,
"learning_rate": 1.9429759623974992e-05,
"loss": 1.109241771697998,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 3.03206,
"step": 80,
"tokens/total": 10485760,
"tokens/train_per_sec_per_gpu": 17934.18,
"tokens/trainable": 10145302
},
{
"epoch": 0.45569620253164556,
"grad_norm": 1.2578125,
"learning_rate": 1.9233550461078114e-05,
"loss": 1.071034049987793,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.9184,
"step": 90,
"tokens/total": 11796480,
"tokens/train_per_sec_per_gpu": 17952.48,
"tokens/trainable": 11415878
},
{
"epoch": 0.5063291139240507,
"grad_norm": 1.2421875,
"learning_rate": 1.900968867902419e-05,
"loss": 1.0816166877746582,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.94944,
"step": 100,
"tokens/total": 13107200,
"tokens/train_per_sec_per_gpu": 17943.02,
"tokens/trainable": 12684573
},
{
"epoch": 0.5063291139240507,
"eval_indic_sft_mini_val_loss": 1.0623797178268433,
"eval_indic_sft_mini_val_runtime": 157.7495,
"eval_indic_sft_mini_val_samples_per_second": 12.171,
"eval_indic_sft_mini_val_steps_per_second": 3.043,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 100
},
{
"epoch": 0.5063291139240507,
"eval_tulu_sft_mini_val_loss": 2.2967746257781982,
"eval_tulu_sft_mini_val_runtime": 92.342,
"eval_tulu_sft_mini_val_samples_per_second": 12.475,
"eval_tulu_sft_mini_val_steps_per_second": 3.119,
"memory/device_reserved (GiB)": 26.58,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 100
},
{
"epoch": 0.5569620253164557,
"grad_norm": 1.2421875,
"learning_rate": 1.8758844698647457e-05,
"loss": 1.106839370727539,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 3.02478,
"step": 110,
"tokens/total": 14417920,
"tokens/train_per_sec_per_gpu": 3955.64,
"tokens/trainable": 13951995
},
{
"epoch": 0.6075949367088608,
"grad_norm": 1.8984375,
"learning_rate": 1.848176974701775e-05,
"loss": 1.0786801338195802,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.9408,
"step": 120,
"tokens/total": 15728640,
"tokens/train_per_sec_per_gpu": 17975.05,
"tokens/trainable": 15221447
},
{
"epoch": 0.6582278481012658,
"grad_norm": 1.28125,
"learning_rate": 1.8179293607667177e-05,
"loss": 1.0739567756652832,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.92694,
"step": 130,
"tokens/total": 17039360,
"tokens/train_per_sec_per_gpu": 17954.31,
"tokens/trainable": 16490903
},
{
"epoch": 0.7088607594936709,
"grad_norm": 1.2890625,
"learning_rate": 1.7852322135555946e-05,
"loss": 1.0759190559387206,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.93269,
"step": 140,
"tokens/total": 18350080,
"tokens/train_per_sec_per_gpu": 17887.36,
"tokens/trainable": 17757184
},
{
"epoch": 0.759493670886076,
"grad_norm": 1.3828125,
"learning_rate": 1.7501834544219697e-05,
"loss": 1.0366504669189454,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.81976,
"step": 150,
"tokens/total": 19660800,
"tokens/train_per_sec_per_gpu": 17911.77,
"tokens/trainable": 19024962
},
{
"epoch": 0.759493670886076,
"eval_indic_sft_mini_val_loss": 1.0453351736068726,
"eval_indic_sft_mini_val_runtime": 157.7623,
"eval_indic_sft_mini_val_samples_per_second": 12.17,
"eval_indic_sft_mini_val_steps_per_second": 3.043,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 150
},
{
"epoch": 0.759493670886076,
"eval_tulu_sft_mini_val_loss": 2.2883615493774414,
"eval_tulu_sft_mini_val_runtime": 92.4616,
"eval_tulu_sft_mini_val_samples_per_second": 12.459,
"eval_tulu_sft_mini_val_steps_per_second": 3.115,
"memory/device_reserved (GiB)": 26.58,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 150
},
{
"epoch": 0.810126582278481,
"grad_norm": 1.328125,
"learning_rate": 1.7128880473222688e-05,
"loss": 1.0812637329101562,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.9484,
"step": 160,
"tokens/total": 20971520,
"tokens/train_per_sec_per_gpu": 3951.59,
"tokens/trainable": 20291324
},
{
"epoch": 0.8607594936708861,
"grad_norm": 1.328125,
"learning_rate": 1.6734576844699234e-05,
"loss": 1.0667606353759767,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.90595,
"step": 170,
"tokens/total": 22282240,
"tokens/train_per_sec_per_gpu": 17990.97,
"tokens/trainable": 21559984
},
{
"epoch": 0.9113924050632911,
"grad_norm": 1.3515625,
"learning_rate": 1.6320104518397473e-05,
"loss": 1.067934513092041,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.90936,
"step": 180,
"tokens/total": 23592960,
"tokens/train_per_sec_per_gpu": 17944.73,
"tokens/trainable": 22828372
},
{
"epoch": 0.9620253164556962,
"grad_norm": 1.296875,
"learning_rate": 1.588670475524283e-05,
"loss": 1.04850492477417,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.85338,
"step": 190,
"tokens/total": 24903680,
"tokens/train_per_sec_per_gpu": 17890.69,
"tokens/trainable": 24094636
},
{
"epoch": 1.010126582278481,
"grad_norm": 1.3046875,
"learning_rate": 1.5435675500012212e-05,
"loss": 1.0491924285888672,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.85534,
"step": 200,
"tokens/total": 26136576,
"tokens/train_per_sec_per_gpu": 17465.18,
"tokens/trainable": 25286404
},
{
"epoch": 1.010126582278481,
"eval_indic_sft_mini_val_loss": 1.0354256629943848,
"eval_indic_sft_mini_val_runtime": 157.741,
"eval_indic_sft_mini_val_samples_per_second": 12.172,
"eval_indic_sft_mini_val_steps_per_second": 3.043,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 200
},
{
"epoch": 1.010126582278481,
"eval_tulu_sft_mini_val_loss": 2.291062593460083,
"eval_tulu_sft_mini_val_runtime": 92.344,
"eval_tulu_sft_mini_val_samples_per_second": 12.475,
"eval_tulu_sft_mini_val_steps_per_second": 3.119,
"memory/device_reserved (GiB)": 26.58,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 200
},
{
"epoch": 1.0607594936708862,
"grad_norm": 1.375,
"learning_rate": 1.4968367494251486e-05,
"loss": 1.0487144470214844,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.85398,
"step": 210,
"tokens/total": 27447296,
"tokens/train_per_sec_per_gpu": 3957.26,
"tokens/trainable": 26554084
},
{
"epoch": 1.111392405063291,
"grad_norm": 1.3671875,
"learning_rate": 1.4486180231077278e-05,
"loss": 1.0355692863464356,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.81671,
"step": 220,
"tokens/total": 28758016,
"tokens/train_per_sec_per_gpu": 17979.8,
"tokens/trainable": 27823280
},
{
"epoch": 1.1620253164556962,
"grad_norm": 1.265625,
"learning_rate": 1.3990557763977694e-05,
"loss": 1.0380287170410156,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.82365,
"step": 230,
"tokens/total": 30068736,
"tokens/train_per_sec_per_gpu": 17964.75,
"tokens/trainable": 29092706
},
{
"epoch": 1.2126582278481013,
"grad_norm": 1.2734375,
"learning_rate": 1.3482984382163713e-05,
"loss": 1.036821174621582,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.82024,
"step": 240,
"tokens/total": 31379456,
"tokens/train_per_sec_per_gpu": 17930.26,
"tokens/trainable": 30360732
},
{
"epoch": 1.2632911392405064,
"grad_norm": 1.2265625,
"learning_rate": 1.2964980165422701e-05,
"loss": 0.995778751373291,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.70683,
"step": 250,
"tokens/total": 32690176,
"tokens/train_per_sec_per_gpu": 17927.94,
"tokens/trainable": 31631352
},
{
"epoch": 1.2632911392405064,
"eval_indic_sft_mini_val_loss": 1.027944803237915,
"eval_indic_sft_mini_val_runtime": 157.8775,
"eval_indic_sft_mini_val_samples_per_second": 12.161,
"eval_indic_sft_mini_val_steps_per_second": 3.04,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 250
},
{
"epoch": 1.2632911392405064,
"eval_tulu_sft_mini_val_loss": 2.2984113693237305,
"eval_tulu_sft_mini_val_runtime": 92.3048,
"eval_tulu_sft_mini_val_samples_per_second": 12.48,
"eval_tulu_sft_mini_val_steps_per_second": 3.12,
"memory/device_reserved (GiB)": 26.58,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 250
},
{
"epoch": 1.3139240506329113,
"grad_norm": 1.2109375,
"learning_rate": 1.2438096431786408e-05,
"loss": 1.0438777923583984,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.84021,
"step": 260,
"tokens/total": 34000896,
"tokens/train_per_sec_per_gpu": 3955.89,
"tokens/trainable": 32899468
},
{
"epoch": 1.3645569620253164,
"grad_norm": 1.28125,
"learning_rate": 1.1903911091646684e-05,
"loss": 1.0240283012390137,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.78439,
"step": 270,
"tokens/total": 35311616,
"tokens/train_per_sec_per_gpu": 17961.08,
"tokens/trainable": 34168384
},
{
"epoch": 1.4151898734177215,
"grad_norm": 1.234375,
"learning_rate": 1.1364023922232503e-05,
"loss": 1.031261920928955,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.8046,
"step": 280,
"tokens/total": 36622336,
"tokens/train_per_sec_per_gpu": 17939.56,
"tokens/trainable": 35436704
},
{
"epoch": 1.4658227848101266,
"grad_norm": 1.3359375,
"learning_rate": 1.0820051776600175e-05,
"loss": 1.0369884490966796,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.82071,
"step": 290,
"tokens/total": 37933056,
"tokens/train_per_sec_per_gpu": 17936.98,
"tokens/trainable": 36705368
},
{
"epoch": 1.5164556962025317,
"grad_norm": 1.1875,
"learning_rate": 1.0273623741484924e-05,
"loss": 1.0238310813903808,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.78384,
"step": 300,
"tokens/total": 39243776,
"tokens/train_per_sec_per_gpu": 17926.64,
"tokens/trainable": 37973400
},
{
"epoch": 1.5164556962025317,
"eval_indic_sft_mini_val_loss": 1.0224663019180298,
"eval_indic_sft_mini_val_runtime": 157.771,
"eval_indic_sft_mini_val_samples_per_second": 12.17,
"eval_indic_sft_mini_val_steps_per_second": 3.042,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 300
},
{
"epoch": 1.5164556962025317,
"eval_tulu_sft_mini_val_loss": 2.295070171356201,
"eval_tulu_sft_mini_val_runtime": 92.349,
"eval_tulu_sft_mini_val_samples_per_second": 12.474,
"eval_tulu_sft_mini_val_steps_per_second": 3.119,
"memory/device_reserved (GiB)": 26.58,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 300
},
{
"epoch": 1.5670886075949366,
"grad_norm": 1.2421875,
"learning_rate": 9.726376258515077e-06,
"loss": 1.0644823074340821,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.89934,
"step": 310,
"tokens/total": 40554496,
"tokens/train_per_sec_per_gpu": 3956.31,
"tokens/trainable": 39240980
},
{
"epoch": 1.6177215189873417,
"grad_norm": 1.234375,
"learning_rate": 9.179948223399828e-06,
"loss": 1.0305088996887206,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.80249,
"step": 320,
"tokens/total": 41865216,
"tokens/train_per_sec_per_gpu": 17965.06,
"tokens/trainable": 40509632
},
{
"epoch": 1.6683544303797468,
"grad_norm": 1.265625,
"learning_rate": 8.6359760777675e-06,
"loss": 1.0250298500061035,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.78718,
"step": 330,
"tokens/total": 43175936,
"tokens/train_per_sec_per_gpu": 17953.88,
"tokens/trainable": 41777768
},
{
"epoch": 1.7189873417721517,
"grad_norm": 1.2265625,
"learning_rate": 8.096088908353316e-06,
"loss": 1.0106207847595214,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.74731,
"step": 340,
"tokens/total": 44486656,
"tokens/train_per_sec_per_gpu": 17933.81,
"tokens/trainable": 43045456
},
{
"epoch": 1.769620253164557,
"grad_norm": 1.2578125,
"learning_rate": 7.561903568213595e-06,
"loss": 0.9956131935119629,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.70638,
"step": 350,
"tokens/total": 45797376,
"tokens/train_per_sec_per_gpu": 17927.21,
"tokens/trainable": 44311528
},
{
"epoch": 1.769620253164557,
"eval_indic_sft_mini_val_loss": 1.0202709436416626,
"eval_indic_sft_mini_val_runtime": 157.7597,
"eval_indic_sft_mini_val_samples_per_second": 12.17,
"eval_indic_sft_mini_val_steps_per_second": 3.043,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 350
},
{
"epoch": 1.769620253164557,
"eval_tulu_sft_mini_val_loss": 2.2989728450775146,
"eval_tulu_sft_mini_val_runtime": 92.4784,
"eval_tulu_sft_mini_val_samples_per_second": 12.457,
"eval_tulu_sft_mini_val_steps_per_second": 3.114,
"memory/device_reserved (GiB)": 26.58,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 350
},
{
"epoch": 1.820253164556962,
"grad_norm": 1.2734375,
"learning_rate": 7.035019834577301e-06,
"loss": 1.0437658309936524,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.83989,
"step": 360,
"tokens/total": 47108096,
"tokens/train_per_sec_per_gpu": 3952.58,
"tokens/trainable": 45578288
},
{
"epoch": 1.870886075949367,
"grad_norm": 1.2265625,
"learning_rate": 6.517015617836292e-06,
"loss": 1.0040513038635255,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.72932,
"step": 370,
"tokens/total": 48418816,
"tokens/train_per_sec_per_gpu": 17939.3,
"tokens/trainable": 46846324
},
{
"epoch": 1.9215189873417722,
"grad_norm": 1.234375,
"learning_rate": 6.009442236022307e-06,
"loss": 1.0151902198791505,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.75989,
"step": 380,
"tokens/total": 49729536,
"tokens/train_per_sec_per_gpu": 17915.04,
"tokens/trainable": 48113428
},
{
"epoch": 1.972151898734177,
"grad_norm": 1.3125,
"learning_rate": 5.513819768922723e-06,
"loss": 0.997506046295166,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.71151,
"step": 390,
"tokens/total": 51040256,
"tokens/train_per_sec_per_gpu": 17963.73,
"tokens/trainable": 49382776
},
{
"epoch": 2.020253164556962,
"grad_norm": 1.2421875,
"learning_rate": 5.031632505748516e-06,
"loss": 1.0011167526245117,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.72132,
"step": 400,
"tokens/total": 52273152,
"tokens/train_per_sec_per_gpu": 17458.66,
"tokens/trainable": 50572704
},
{
"epoch": 2.020253164556962,
"eval_indic_sft_mini_val_loss": 1.0187087059020996,
"eval_indic_sft_mini_val_runtime": 158.0242,
"eval_indic_sft_mini_val_samples_per_second": 12.15,
"eval_indic_sft_mini_val_steps_per_second": 3.038,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 400
},
{
"epoch": 2.020253164556962,
"eval_tulu_sft_mini_val_loss": 2.2966930866241455,
"eval_tulu_sft_mini_val_runtime": 92.3222,
"eval_tulu_sft_mini_val_samples_per_second": 12.478,
"eval_tulu_sft_mini_val_steps_per_second": 3.12,
"memory/device_reserved (GiB)": 26.58,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 400
},
{
"epoch": 2.070886075949367,
"grad_norm": 1.2421875,
"learning_rate": 4.56432449998779e-06,
"loss": 1.0103323936462403,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.74651,
"step": 410,
"tokens/total": 53583872,
"tokens/train_per_sec_per_gpu": 3952.63,
"tokens/trainable": 51839796
},
{
"epoch": 2.1215189873417724,
"grad_norm": 1.234375,
"learning_rate": 4.113295244757171e-06,
"loss": 1.0133058547973632,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.75469,
"step": 420,
"tokens/total": 54894592,
"tokens/train_per_sec_per_gpu": 17984.79,
"tokens/trainable": 53108816
},
{
"epoch": 2.1721518987341772,
"grad_norm": 1.21875,
"learning_rate": 3.679895481602529e-06,
"loss": 1.0214984893798829,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.77735,
"step": 430,
"tokens/total": 56205312,
"tokens/train_per_sec_per_gpu": 17959.57,
"tokens/trainable": 54378168
},
{
"epoch": 2.222784810126582,
"grad_norm": 1.2109375,
"learning_rate": 3.2654231553007665e-06,
"loss": 0.9840593338012695,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.67529,
"step": 440,
"tokens/total": 57516032,
"tokens/train_per_sec_per_gpu": 17930.45,
"tokens/trainable": 55645416
},
{
"epoch": 2.2734177215189875,
"grad_norm": 1.1953125,
"learning_rate": 2.871119526777315e-06,
"loss": 1.028823184967041,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.79777,
"step": 450,
"tokens/total": 58826752,
"tokens/train_per_sec_per_gpu": 17944.17,
"tokens/trainable": 56914256
},
{
"epoch": 2.2734177215189875,
"eval_indic_sft_mini_val_loss": 1.0182074308395386,
"eval_indic_sft_mini_val_runtime": 157.9039,
"eval_indic_sft_mini_val_samples_per_second": 12.159,
"eval_indic_sft_mini_val_steps_per_second": 3.04,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 450
},
{
"epoch": 2.2734177215189875,
"eval_tulu_sft_mini_val_loss": 2.297419309616089,
"eval_tulu_sft_mini_val_runtime": 92.4484,
"eval_tulu_sft_mini_val_samples_per_second": 12.461,
"eval_tulu_sft_mini_val_steps_per_second": 3.115,
"memory/device_reserved (GiB)": 26.59,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 450
},
{
"epoch": 2.3240506329113924,
"grad_norm": 1.21875,
"learning_rate": 2.4981654557803026e-06,
"loss": 1.0112363815307617,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.749,
"step": 460,
"tokens/total": 60137472,
"tokens/train_per_sec_per_gpu": 3954.28,
"tokens/trainable": 58182604
},
{
"epoch": 2.3746835443037977,
"grad_norm": 1.265625,
"learning_rate": 2.1476778644440553e-06,
"loss": 0.9997093200683593,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.71749,
"step": 470,
"tokens/total": 61448192,
"tokens/train_per_sec_per_gpu": 17956.34,
"tokens/trainable": 59450748
},
{
"epoch": 2.4253164556962026,
"grad_norm": 1.21875,
"learning_rate": 1.820706392332824e-06,
"loss": 1.0180435180664062,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.76777,
"step": 480,
"tokens/total": 62758912,
"tokens/train_per_sec_per_gpu": 17939.84,
"tokens/trainable": 60720528
},
{
"epoch": 2.4759493670886075,
"grad_norm": 1.234375,
"learning_rate": 1.518230252982248e-06,
"loss": 1.040645217895508,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.83104,
"step": 490,
"tokens/total": 64069632,
"tokens/train_per_sec_per_gpu": 17931.8,
"tokens/trainable": 61988728
},
{
"epoch": 2.526582278481013,
"grad_norm": 1.171875,
"learning_rate": 1.2411553013525457e-06,
"loss": 1.0239150047302246,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 32.29,
"memory/max_allocated (GiB)": 32.29,
"ppl": 2.78407,
"step": 500,
"tokens/total": 65380352,
"tokens/train_per_sec_per_gpu": 17916.35,
"tokens/trainable": 63254896
},
{
"epoch": 2.526582278481013,
"eval_indic_sft_mini_val_loss": 1.018338680267334,
"eval_indic_sft_mini_val_runtime": 157.6995,
"eval_indic_sft_mini_val_samples_per_second": 12.175,
"eval_indic_sft_mini_val_steps_per_second": 3.044,
"memory/device_reserved (GiB)": 37.04,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 500
},
{
"epoch": 2.526582278481013,
"eval_tulu_sft_mini_val_loss": 2.299147605895996,
"eval_tulu_sft_mini_val_runtime": 92.1488,
"eval_tulu_sft_mini_val_samples_per_second": 12.502,
"eval_tulu_sft_mini_val_steps_per_second": 3.125,
"memory/device_reserved (GiB)": 26.59,
"memory/max_active (GiB)": 25.99,
"memory/max_allocated (GiB)": 25.99,
"step": 500
}
],
"logging_steps": 10,
"max_steps": 591,
"num_input_tokens_seen": 0,
"num_train_epochs": 3,
"save_steps": 500,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 1.4039702725617254e+17,
"train_batch_size": 4,
"trial_name": null,
"trial_params": null
}

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:f1ab83fb7a53e79e01752f8af988eed016a56cd8159a16af0f41fcfaaa076803
size 9169

View File

@@ -0,0 +1,54 @@
{%- if tools %}
{{- '<|im_start|>system\n' }}
{%- if messages[0]['role'] == 'system' %}
{{- messages[0]['content'] }}
{%- else %}
{{- 'You are a helpful assistant.' }}
{%- endif %}
{{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
{%- for tool in tools %}
{{- "\n" }}
{{- tool | tojson }}
{%- endfor %}
{{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
{%- else %}
{%- if messages[0]['role'] == 'system' %}
{{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
{%- else %}
{{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
{%- endif %}
{%- endif %}
{%- for message in messages %}
{%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
{{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
{%- elif message.role == "assistant" %}
{{- '<|im_start|>' + message.role }}
{%- if message.content %}
{{- '\n' + message.content }}
{%- endif %}
{%- for tool_call in message.tool_calls %}
{%- if tool_call.function is defined %}
{%- set tool_call = tool_call.function %}
{%- endif %}
{{- '\n<tool_call>\n{"name": "' }}
{{- tool_call.name }}
{{- '", "arguments": ' }}
{{- tool_call.arguments | tojson }}
{{- '}\n</tool_call>' }}
{%- endfor %}
{{- '<|im_end|>\n' }}
{%- elif message.role == "tool" %}
{%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
{{- '<|im_start|>user' }}
{%- endif %}
{{- '\n<tool_response>\n' }}
{{- message.content }}
{{- '\n</tool_response>' }}
{%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
{{- '<|im_end|>\n' }}
{%- endif %}
{%- endif %}
{%- endfor %}
{%- if add_generation_prompt %}
{{- '<|im_start|>assistant\n' }}
{%- endif %}

View File

@@ -0,0 +1,58 @@
{
"architectures": [
"Qwen2ForCausalLM"
],
"attention_dropout": 0.0,
"bos_token_id": null,
"dtype": "bfloat16",
"eos_token_id": 151643,
"hidden_act": "silu",
"hidden_size": 896,
"initializer_range": 0.02,
"intermediate_size": 4864,
"layer_types": [
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention"
],
"max_position_embeddings": 32768,
"max_window_layers": 24,
"model_type": "qwen2",
"num_attention_heads": 14,
"num_hidden_layers": 24,
"num_key_value_heads": 2,
"pad_token_id": 151643,
"rms_norm_eps": 1e-06,
"rope_parameters": {
"rope_theta": 1000000.0,
"rope_type": "default"
},
"sliding_window": null,
"tie_word_embeddings": true,
"transformers_version": "5.14.1",
"use_cache": false,
"use_mrope": false,
"use_sliding_window": false,
"vocab_size": 151936
}

View File

@@ -0,0 +1,9 @@
{
"do_sample": true,
"eos_token_id": [
151643
],
"max_new_tokens": 2048,
"pad_token_id": 151643,
"transformers_version": "5.14.1"
}

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:4f2ef297270ca28cd4a413f624c81a6c0e639cc6667fbbfd6fdab6640a114eb6
size 988097824

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:3fd169731d2cbde95e10bf356d66d5997fd885dd8dbb6fb4684da3f23b2585d8
size 11421892

View File

@@ -0,0 +1,30 @@
{
"add_prefix_space": false,
"backend": "tokenizers",
"bos_token": null,
"clean_up_tokenization_spaces": false,
"eos_token": "<|endoftext|>",
"errors": "replace",
"extra_special_tokens": [
"<|im_start|>",
"<|im_end|>",
"<|object_ref_start|>",
"<|object_ref_end|>",
"<|box_start|>",
"<|box_end|>",
"<|quad_start|>",
"<|quad_end|>",
"<|vision_start|>",
"<|vision_end|>",
"<|vision_pad|>",
"<|image_pad|>",
"<|video_pad|>"
],
"is_local": false,
"local_files_only": false,
"model_max_length": 131072,
"pad_token": "<|endoftext|>",
"split_special_tokens": false,
"tokenizer_class": "Qwen2Tokenizer",
"unk_token": null
}

View File

@@ -0,0 +1 @@
{"total": 77307904, "trainable": 74795984}

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:f1ab83fb7a53e79e01752f8af988eed016a56cd8159a16af0f41fcfaaa076803
size 9169

58
config.json Normal file
View File

@@ -0,0 +1,58 @@
{
"architectures": [
"Qwen2ForCausalLM"
],
"attention_dropout": 0.0,
"bos_token_id": null,
"dtype": "bfloat16",
"eos_token_id": 151643,
"hidden_act": "silu",
"hidden_size": 896,
"initializer_range": 0.02,
"intermediate_size": 4864,
"layer_types": [
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention"
],
"max_position_embeddings": 32768,
"max_window_layers": 24,
"model_type": "qwen2",
"num_attention_heads": 14,
"num_hidden_layers": 24,
"num_key_value_heads": 2,
"pad_token_id": 151643,
"rms_norm_eps": 1e-06,
"rope_parameters": {
"rope_theta": 1000000.0,
"rope_type": "default"
},
"sliding_window": null,
"tie_word_embeddings": true,
"transformers_version": "5.14.1",
"use_cache": false,
"use_mrope": false,
"use_sliding_window": false,
"vocab_size": 151936
}

10560
debug.log Normal file

File diff suppressed because it is too large Load Diff

9
generation_config.json Normal file
View File

@@ -0,0 +1,9 @@
{
"do_sample": true,
"eos_token_id": [
151643
],
"max_new_tokens": 2048,
"pad_token_id": 151643,
"transformers_version": "5.14.1"
}

3
model.safetensors Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:4f2ef297270ca28cd4a413f624c81a6c0e639cc6667fbbfd6fdab6640a114eb6
size 988097824

3
tokenizer.json Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:3fd169731d2cbde95e10bf356d66d5997fd885dd8dbb6fb4684da3f23b2585d8
size 11421892

30
tokenizer_config.json Normal file
View File

@@ -0,0 +1,30 @@
{
"add_prefix_space": false,
"backend": "tokenizers",
"bos_token": null,
"clean_up_tokenization_spaces": false,
"eos_token": "<|endoftext|>",
"errors": "replace",
"extra_special_tokens": [
"<|im_start|>",
"<|im_end|>",
"<|object_ref_start|>",
"<|object_ref_end|>",
"<|box_start|>",
"<|box_end|>",
"<|quad_start|>",
"<|quad_end|>",
"<|vision_start|>",
"<|vision_end|>",
"<|vision_pad|>",
"<|image_pad|>",
"<|video_pad|>"
],
"is_local": false,
"local_files_only": false,
"model_max_length": 131072,
"pad_token": "<|endoftext|>",
"split_special_tokens": false,
"tokenizer_class": "Qwen2Tokenizer",
"unk_token": null
}

3
training_args.bin Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:f1ab83fb7a53e79e01752f8af988eed016a56cd8159a16af0f41fcfaaa076803
size 9169