初始化项目,由ModelHub XC社区提供模型
Model: adityabanerjee13/qwen2.5-0.5b-sft-IT Source: Original Platform
This commit is contained in:
36
.gitattributes
vendored
Normal file
36
.gitattributes
vendored
Normal file
@@ -0,0 +1,36 @@
|
||||
*.7z filter=lfs diff=lfs merge=lfs -text
|
||||
*.arrow filter=lfs diff=lfs merge=lfs -text
|
||||
*.bin filter=lfs diff=lfs merge=lfs -text
|
||||
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
||||
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
||||
*.ftz filter=lfs diff=lfs merge=lfs -text
|
||||
*.gz filter=lfs diff=lfs merge=lfs -text
|
||||
*.h5 filter=lfs diff=lfs merge=lfs -text
|
||||
*.joblib filter=lfs diff=lfs merge=lfs -text
|
||||
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
||||
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
||||
*.model filter=lfs diff=lfs merge=lfs -text
|
||||
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
||||
*.npy filter=lfs diff=lfs merge=lfs -text
|
||||
*.npz filter=lfs diff=lfs merge=lfs -text
|
||||
*.onnx filter=lfs diff=lfs merge=lfs -text
|
||||
*.ot filter=lfs diff=lfs merge=lfs -text
|
||||
*.parquet filter=lfs diff=lfs merge=lfs -text
|
||||
*.pb filter=lfs diff=lfs merge=lfs -text
|
||||
*.pickle filter=lfs diff=lfs merge=lfs -text
|
||||
*.pkl filter=lfs diff=lfs merge=lfs -text
|
||||
*.pt filter=lfs diff=lfs merge=lfs -text
|
||||
*.pth filter=lfs diff=lfs merge=lfs -text
|
||||
*.rar filter=lfs diff=lfs merge=lfs -text
|
||||
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
||||
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
||||
*.tar filter=lfs diff=lfs merge=lfs -text
|
||||
*.tflite filter=lfs diff=lfs merge=lfs -text
|
||||
*.tgz filter=lfs diff=lfs merge=lfs -text
|
||||
*.wasm filter=lfs diff=lfs merge=lfs -text
|
||||
*.xz filter=lfs diff=lfs merge=lfs -text
|
||||
*.zip filter=lfs diff=lfs merge=lfs -text
|
||||
*.zst filter=lfs diff=lfs merge=lfs -text
|
||||
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
||||
tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
||||
179
README.md
Normal file
179
README.md
Normal file
@@ -0,0 +1,179 @@
|
||||
---
|
||||
library_name: transformers
|
||||
license: apache-2.0
|
||||
base_model: adityabanerjee13/qwen2.5-0.5b-cpt-mix-1to2
|
||||
tags:
|
||||
- axolotl
|
||||
- generated_from_trainer
|
||||
datasets:
|
||||
- adityabanerjee13/indic-sft-mini-train
|
||||
- adityabanerjee13/tulu-sft-mini-train
|
||||
model-index:
|
||||
- name: qwen2.5-0.5b-sft-IT
|
||||
results: []
|
||||
---
|
||||
|
||||
<!-- This model card has been generated automatically according to the information the Trainer had access to. You
|
||||
should probably proofread and complete it, then remove this comment. -->
|
||||
|
||||
[<img src="https://raw.githubusercontent.com/axolotl-ai-cloud/axolotl/main/image/axolotl-badge-web.png" alt="Built with Axolotl" width="200" height="32"/>](https://github.com/axolotl-ai-cloud/axolotl)
|
||||
<details><summary>See axolotl config</summary>
|
||||
|
||||
axolotl version: `0.19.0.dev0`
|
||||
```yaml
|
||||
# ==============================================================================
|
||||
# Axolotl CPT config — Qwen2.5-0.5B, full-parameter, single GPU.
|
||||
# Data mix RATIO EXPERIMENT (character-level exact):
|
||||
#
|
||||
# RUN 2 of 3 — fineweb : indic = 1 : 2 (FineWeb is HALF the Indic size)
|
||||
# FineWeb web-crawl chars == Indic train chars / 2.
|
||||
#
|
||||
# Indic train : adityabanerjee13/indic-cpt-mini-train (7,907,882 chars)
|
||||
# FineWeb train: adityabanerjee13/fineweb-cpt-half (3,953,941 chars)
|
||||
# Validation : adityabanerjee13/indic-cpt-mini-val (held-out 1% Indic)
|
||||
#
|
||||
# The datasets are pre-sized to exact character counts on the Hub, so loading
|
||||
# each one whole gives the exact 1:2 ratio — no slicing needed.
|
||||
#
|
||||
# Usage:
|
||||
# python train.py --config qwen2.5_0.5b_cpt_mix_1to2.yml
|
||||
# ==============================================================================
|
||||
|
||||
base_model: adityabanerjee13/qwen2.5-0.5b-cpt-mix-1to2
|
||||
model_type: AutoModelForCausalLM
|
||||
tokenizer_type: AutoTokenizer
|
||||
trust_remote_code: false
|
||||
|
||||
adapter:
|
||||
load_in_8bit: false
|
||||
load_in_4bit: false
|
||||
|
||||
# multi_eval_plugin splits test_datasets back into per-source eval sets so
|
||||
# this run logs eval_indic_cpt_mini_val_loss and eval_fineweb_cpt_val_loss
|
||||
# separately (instead of one merged eval_loss) at every eval step, incl. to
|
||||
# wandb. Requires this folder on PYTHONPATH — launch via
|
||||
# `python train.py --config <this file>`.
|
||||
plugins:
|
||||
- multi_eval_plugin.MultiEvalPlugin
|
||||
|
||||
datasets:
|
||||
- path: adityabanerjee13/indic-sft-mini-train
|
||||
type: chat_template
|
||||
field_messages: messages
|
||||
split: train
|
||||
- path: adityabanerjee13/tulu-sft-mini-train
|
||||
type: chat_template
|
||||
field_messages: messages
|
||||
split: train
|
||||
|
||||
test_datasets:
|
||||
- path: adityabanerjee13/indic-sft-mini-val
|
||||
type: chat_template
|
||||
field_messages: messages
|
||||
split: validation
|
||||
- path: adityabanerjee13/tulu-sft-mini-val
|
||||
type: chat_template
|
||||
field_messages: messages
|
||||
split: validation
|
||||
|
||||
train_on_inputs: false
|
||||
|
||||
chat_template: tokenizer_default
|
||||
|
||||
dataset_prepared_path: ./last_run_prepared_1to2
|
||||
dataset_num_proc: 1 # single-process tokenize: avoids fork deadlock
|
||||
val_set_size: 0
|
||||
output_dir: ./outputs/qwen2.5-0.5b-sft-IT
|
||||
|
||||
# --- Sequence packing -----------------------------------------------------
|
||||
sequence_len: 4096
|
||||
sample_packing: true
|
||||
pad_to_sequence_len: true
|
||||
eval_sample_packing: false
|
||||
|
||||
# --- Optimization ----------------------------------------------------------
|
||||
gradient_accumulation_steps: 8
|
||||
micro_batch_size: 4
|
||||
num_epochs: 3
|
||||
optimizer: adamw_torch_fused
|
||||
lr_scheduler: cosine
|
||||
learning_rate: 2e-5
|
||||
warmup_ratio: 0.03
|
||||
weight_decay: 0.01
|
||||
max_grad_norm: 1.0
|
||||
|
||||
train_on_inputs: true
|
||||
group_by_length: false
|
||||
|
||||
# --- Precision / memory ---------------------------------------------------
|
||||
bf16: auto
|
||||
fp16:
|
||||
tf32: true
|
||||
gradient_checkpointing: true
|
||||
flash_attention: true
|
||||
|
||||
# --- Logging / checkpoints ------------------------------------------------
|
||||
logging_steps: 10
|
||||
save_strategy: steps
|
||||
save_steps: 500
|
||||
save_total_limit: 30
|
||||
save_only_model: true # save weights only — no optimizer/scheduler state
|
||||
# (checkpoints ~1/3 the size; can't resume training)
|
||||
evals_per_epoch: 4
|
||||
|
||||
wandb_project: indic-sft
|
||||
wandb_entity: models-na9841
|
||||
wandb_name: qwen2.5-0.5b-sft-IT
|
||||
wandb_log_model: "false"
|
||||
|
||||
hub_model_id: adityabanerjee13/qwen2.5-0.5b-sft-IT
|
||||
hub_strategy: all_checkpoints
|
||||
|
||||
special_tokens:
|
||||
|
||||
```
|
||||
|
||||
</details><br>
|
||||
|
||||
# qwen2.5-0.5b-sft-IT
|
||||
|
||||
This model is a fine-tuned version of [adityabanerjee13/qwen2.5-0.5b-cpt-mix-1to2](https://huggingface.co/adityabanerjee13/qwen2.5-0.5b-cpt-mix-1to2) on the adityabanerjee13/indic-sft-mini-train and the adityabanerjee13/tulu-sft-mini-train datasets.
|
||||
|
||||
## Model description
|
||||
|
||||
More information needed
|
||||
|
||||
## Intended uses & limitations
|
||||
|
||||
More information needed
|
||||
|
||||
## Training and evaluation data
|
||||
|
||||
More information needed
|
||||
|
||||
## Training procedure
|
||||
|
||||
### Training hyperparameters
|
||||
|
||||
The following hyperparameters were used during training:
|
||||
- learning_rate: 2e-05
|
||||
- train_batch_size: 4
|
||||
- eval_batch_size: 4
|
||||
- seed: 42
|
||||
- gradient_accumulation_steps: 8
|
||||
- total_train_batch_size: 32
|
||||
- optimizer: Use OptimizerNames.ADAMW_TORCH_FUSED with betas=(0.9,0.999) and epsilon=1e-08 and optimizer_args=No additional optimizer arguments
|
||||
- lr_scheduler_type: cosine
|
||||
- lr_scheduler_warmup_steps: 17
|
||||
- training_steps: 591
|
||||
|
||||
### Training results
|
||||
|
||||
|
||||
|
||||
### Framework versions
|
||||
|
||||
- Transformers 5.14.1
|
||||
- Pytorch 2.12.0+cu130
|
||||
- Datasets 4.8.4
|
||||
- Tokenizers 0.22.2
|
||||
54
chat_template.jinja
Normal file
54
chat_template.jinja
Normal file
@@ -0,0 +1,54 @@
|
||||
{%- if tools %}
|
||||
{{- '<|im_start|>system\n' }}
|
||||
{%- if messages[0]['role'] == 'system' %}
|
||||
{{- messages[0]['content'] }}
|
||||
{%- else %}
|
||||
{{- 'You are a helpful assistant.' }}
|
||||
{%- endif %}
|
||||
{{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
|
||||
{%- for tool in tools %}
|
||||
{{- "\n" }}
|
||||
{{- tool | tojson }}
|
||||
{%- endfor %}
|
||||
{{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
|
||||
{%- else %}
|
||||
{%- if messages[0]['role'] == 'system' %}
|
||||
{{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
|
||||
{%- else %}
|
||||
{{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
|
||||
{%- endif %}
|
||||
{%- endif %}
|
||||
{%- for message in messages %}
|
||||
{%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
|
||||
{{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
|
||||
{%- elif message.role == "assistant" %}
|
||||
{{- '<|im_start|>' + message.role }}
|
||||
{%- if message.content %}
|
||||
{{- '\n' + message.content }}
|
||||
{%- endif %}
|
||||
{%- for tool_call in message.tool_calls %}
|
||||
{%- if tool_call.function is defined %}
|
||||
{%- set tool_call = tool_call.function %}
|
||||
{%- endif %}
|
||||
{{- '\n<tool_call>\n{"name": "' }}
|
||||
{{- tool_call.name }}
|
||||
{{- '", "arguments": ' }}
|
||||
{{- tool_call.arguments | tojson }}
|
||||
{{- '}\n</tool_call>' }}
|
||||
{%- endfor %}
|
||||
{{- '<|im_end|>\n' }}
|
||||
{%- elif message.role == "tool" %}
|
||||
{%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
|
||||
{{- '<|im_start|>user' }}
|
||||
{%- endif %}
|
||||
{{- '\n<tool_response>\n' }}
|
||||
{{- message.content }}
|
||||
{{- '\n</tool_response>' }}
|
||||
{%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
|
||||
{{- '<|im_end|>\n' }}
|
||||
{%- endif %}
|
||||
{%- endif %}
|
||||
{%- endfor %}
|
||||
{%- if add_generation_prompt %}
|
||||
{{- '<|im_start|>assistant\n' }}
|
||||
{%- endif %}
|
||||
54
checkpoint-500/chat_template.jinja
Normal file
54
checkpoint-500/chat_template.jinja
Normal file
@@ -0,0 +1,54 @@
|
||||
{%- if tools %}
|
||||
{{- '<|im_start|>system\n' }}
|
||||
{%- if messages[0]['role'] == 'system' %}
|
||||
{{- messages[0]['content'] }}
|
||||
{%- else %}
|
||||
{{- 'You are a helpful assistant.' }}
|
||||
{%- endif %}
|
||||
{{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
|
||||
{%- for tool in tools %}
|
||||
{{- "\n" }}
|
||||
{{- tool | tojson }}
|
||||
{%- endfor %}
|
||||
{{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
|
||||
{%- else %}
|
||||
{%- if messages[0]['role'] == 'system' %}
|
||||
{{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
|
||||
{%- else %}
|
||||
{{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
|
||||
{%- endif %}
|
||||
{%- endif %}
|
||||
{%- for message in messages %}
|
||||
{%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
|
||||
{{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
|
||||
{%- elif message.role == "assistant" %}
|
||||
{{- '<|im_start|>' + message.role }}
|
||||
{%- if message.content %}
|
||||
{{- '\n' + message.content }}
|
||||
{%- endif %}
|
||||
{%- for tool_call in message.tool_calls %}
|
||||
{%- if tool_call.function is defined %}
|
||||
{%- set tool_call = tool_call.function %}
|
||||
{%- endif %}
|
||||
{{- '\n<tool_call>\n{"name": "' }}
|
||||
{{- tool_call.name }}
|
||||
{{- '", "arguments": ' }}
|
||||
{{- tool_call.arguments | tojson }}
|
||||
{{- '}\n</tool_call>' }}
|
||||
{%- endfor %}
|
||||
{{- '<|im_end|>\n' }}
|
||||
{%- elif message.role == "tool" %}
|
||||
{%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
|
||||
{{- '<|im_start|>user' }}
|
||||
{%- endif %}
|
||||
{{- '\n<tool_response>\n' }}
|
||||
{{- message.content }}
|
||||
{{- '\n</tool_response>' }}
|
||||
{%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
|
||||
{{- '<|im_end|>\n' }}
|
||||
{%- endif %}
|
||||
{%- endif %}
|
||||
{%- endfor %}
|
||||
{%- if add_generation_prompt %}
|
||||
{{- '<|im_start|>assistant\n' }}
|
||||
{%- endif %}
|
||||
58
checkpoint-500/config.json
Normal file
58
checkpoint-500/config.json
Normal file
@@ -0,0 +1,58 @@
|
||||
{
|
||||
"architectures": [
|
||||
"Qwen2ForCausalLM"
|
||||
],
|
||||
"attention_dropout": 0.0,
|
||||
"bos_token_id": null,
|
||||
"dtype": "bfloat16",
|
||||
"eos_token_id": 151643,
|
||||
"hidden_act": "silu",
|
||||
"hidden_size": 896,
|
||||
"initializer_range": 0.02,
|
||||
"intermediate_size": 4864,
|
||||
"layer_types": [
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention"
|
||||
],
|
||||
"max_position_embeddings": 32768,
|
||||
"max_window_layers": 24,
|
||||
"model_type": "qwen2",
|
||||
"num_attention_heads": 14,
|
||||
"num_hidden_layers": 24,
|
||||
"num_key_value_heads": 2,
|
||||
"pad_token_id": 151643,
|
||||
"rms_norm_eps": 1e-06,
|
||||
"rope_parameters": {
|
||||
"rope_theta": 1000000.0,
|
||||
"rope_type": "default"
|
||||
},
|
||||
"sliding_window": null,
|
||||
"tie_word_embeddings": true,
|
||||
"transformers_version": "5.14.1",
|
||||
"use_cache": false,
|
||||
"use_mrope": false,
|
||||
"use_sliding_window": false,
|
||||
"vocab_size": 151936
|
||||
}
|
||||
9
checkpoint-500/generation_config.json
Normal file
9
checkpoint-500/generation_config.json
Normal file
@@ -0,0 +1,9 @@
|
||||
{
|
||||
"do_sample": true,
|
||||
"eos_token_id": [
|
||||
151643
|
||||
],
|
||||
"max_new_tokens": 2048,
|
||||
"pad_token_id": 151643,
|
||||
"transformers_version": "5.14.1"
|
||||
}
|
||||
3
checkpoint-500/model.safetensors
Normal file
3
checkpoint-500/model.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:9df9ccbbb1556eb3db2d7023d7cef1f34ce7492419e30bad0da7dcb729885eab
|
||||
size 988097824
|
||||
3
checkpoint-500/tokenizer.json
Normal file
3
checkpoint-500/tokenizer.json
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:3fd169731d2cbde95e10bf356d66d5997fd885dd8dbb6fb4684da3f23b2585d8
|
||||
size 11421892
|
||||
30
checkpoint-500/tokenizer_config.json
Normal file
30
checkpoint-500/tokenizer_config.json
Normal file
@@ -0,0 +1,30 @@
|
||||
{
|
||||
"add_prefix_space": false,
|
||||
"backend": "tokenizers",
|
||||
"bos_token": null,
|
||||
"clean_up_tokenization_spaces": false,
|
||||
"eos_token": "<|endoftext|>",
|
||||
"errors": "replace",
|
||||
"extra_special_tokens": [
|
||||
"<|im_start|>",
|
||||
"<|im_end|>",
|
||||
"<|object_ref_start|>",
|
||||
"<|object_ref_end|>",
|
||||
"<|box_start|>",
|
||||
"<|box_end|>",
|
||||
"<|quad_start|>",
|
||||
"<|quad_end|>",
|
||||
"<|vision_start|>",
|
||||
"<|vision_end|>",
|
||||
"<|vision_pad|>",
|
||||
"<|image_pad|>",
|
||||
"<|video_pad|>"
|
||||
],
|
||||
"is_local": false,
|
||||
"local_files_only": false,
|
||||
"model_max_length": 131072,
|
||||
"pad_token": "<|endoftext|>",
|
||||
"split_special_tokens": false,
|
||||
"tokenizer_class": "Qwen2Tokenizer",
|
||||
"unk_token": null
|
||||
}
|
||||
1
checkpoint-500/tokens_state.json
Normal file
1
checkpoint-500/tokens_state.json
Normal file
@@ -0,0 +1 @@
|
||||
{"total": 65380352, "trainable": 63254896}
|
||||
975
checkpoint-500/trainer_state.json
Normal file
975
checkpoint-500/trainer_state.json
Normal file
@@ -0,0 +1,975 @@
|
||||
{
|
||||
"best_global_step": null,
|
||||
"best_metric": null,
|
||||
"best_model_checkpoint": null,
|
||||
"epoch": 2.526582278481013,
|
||||
"eval_steps": 50,
|
||||
"global_step": 500,
|
||||
"is_hyper_param_search": false,
|
||||
"is_local_process_zero": true,
|
||||
"is_world_process_zero": true,
|
||||
"log_history": [
|
||||
{
|
||||
"epoch": 0,
|
||||
"eval_indic_sft_mini_val_loss": 1.127184510231018,
|
||||
"eval_indic_sft_mini_val_runtime": 156.0033,
|
||||
"eval_indic_sft_mini_val_samples_per_second": 12.307,
|
||||
"eval_indic_sft_mini_val_steps_per_second": 3.077,
|
||||
"memory/device_reserved (GiB)": 24.22,
|
||||
"memory/max_active (GiB)": 24.14,
|
||||
"memory/max_allocated (GiB)": 24.14,
|
||||
"step": 0
|
||||
},
|
||||
{
|
||||
"epoch": 0,
|
||||
"eval_tulu_sft_mini_val_loss": 2.330348491668701,
|
||||
"eval_tulu_sft_mini_val_runtime": 91.8069,
|
||||
"eval_tulu_sft_mini_val_samples_per_second": 12.548,
|
||||
"eval_tulu_sft_mini_val_steps_per_second": 3.137,
|
||||
"memory/device_reserved (GiB)": 24.22,
|
||||
"memory/max_active (GiB)": 24.14,
|
||||
"memory/max_allocated (GiB)": 24.14,
|
||||
"step": 0
|
||||
},
|
||||
{
|
||||
"epoch": 0.05063291139240506,
|
||||
"grad_norm": 1.9375,
|
||||
"learning_rate": 1.0588235294117648e-05,
|
||||
"loss": 1.2239381790161132,
|
||||
"memory/device_reserved (GiB)": 37.02,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 3.40055,
|
||||
"step": 10,
|
||||
"tokens/total": 1310720,
|
||||
"tokens/trainable": 1268231
|
||||
},
|
||||
{
|
||||
"epoch": 0.10126582278481013,
|
||||
"grad_norm": 1.640625,
|
||||
"learning_rate": 1.9999400896826965e-05,
|
||||
"loss": 1.1529170989990234,
|
||||
"memory/device_reserved (GiB)": 37.02,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 3.16742,
|
||||
"step": 20,
|
||||
"tokens/total": 2621440,
|
||||
"tokens/train_per_sec_per_gpu": 17988.68,
|
||||
"tokens/trainable": 2535045
|
||||
},
|
||||
{
|
||||
"epoch": 0.1518987341772152,
|
||||
"grad_norm": 1.4453125,
|
||||
"learning_rate": 1.9978439822224228e-05,
|
||||
"loss": 1.1212153434753418,
|
||||
"memory/device_reserved (GiB)": 37.02,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 3.06858,
|
||||
"step": 30,
|
||||
"tokens/total": 3932160,
|
||||
"tokens/train_per_sec_per_gpu": 17958.15,
|
||||
"tokens/trainable": 3804033
|
||||
},
|
||||
{
|
||||
"epoch": 0.20253164556962025,
|
||||
"grad_norm": 1.3828125,
|
||||
"learning_rate": 1.9927595335238736e-05,
|
||||
"loss": 1.106326198577881,
|
||||
"memory/device_reserved (GiB)": 37.02,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 3.02323,
|
||||
"step": 40,
|
||||
"tokens/total": 5242880,
|
||||
"tokens/train_per_sec_per_gpu": 17929.18,
|
||||
"tokens/trainable": 5070883
|
||||
},
|
||||
{
|
||||
"epoch": 0.25316455696202533,
|
||||
"grad_norm": 1.421875,
|
||||
"learning_rate": 1.984701970484229e-05,
|
||||
"loss": 1.0841044425964355,
|
||||
"memory/device_reserved (GiB)": 37.02,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.95679,
|
||||
"step": 50,
|
||||
"tokens/total": 6553600,
|
||||
"tokens/train_per_sec_per_gpu": 17943.82,
|
||||
"tokens/trainable": 6341109
|
||||
},
|
||||
{
|
||||
"epoch": 0.25316455696202533,
|
||||
"eval_indic_sft_mini_val_loss": 1.0926785469055176,
|
||||
"eval_indic_sft_mini_val_runtime": 157.8472,
|
||||
"eval_indic_sft_mini_val_samples_per_second": 12.164,
|
||||
"eval_indic_sft_mini_val_steps_per_second": 3.041,
|
||||
"memory/device_reserved (GiB)": 37.02,
|
||||
"memory/max_active (GiB)": 25.99,
|
||||
"memory/max_allocated (GiB)": 25.99,
|
||||
"step": 50
|
||||
},
|
||||
{
|
||||
"epoch": 0.25316455696202533,
|
||||
"eval_tulu_sft_mini_val_loss": 2.3083696365356445,
|
||||
"eval_tulu_sft_mini_val_runtime": 92.1987,
|
||||
"eval_tulu_sft_mini_val_samples_per_second": 12.495,
|
||||
"eval_tulu_sft_mini_val_steps_per_second": 3.124,
|
||||
"memory/device_reserved (GiB)": 26.58,
|
||||
"memory/max_active (GiB)": 25.99,
|
||||
"memory/max_allocated (GiB)": 25.99,
|
||||
"step": 50
|
||||
},
|
||||
{
|
||||
"epoch": 0.3037974683544304,
|
||||
"grad_norm": 1.34375,
|
||||
"learning_rate": 1.9736954238777793e-05,
|
||||
"loss": 1.1303999900817872,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 3.09689,
|
||||
"step": 60,
|
||||
"tokens/total": 7864320,
|
||||
"tokens/train_per_sec_per_gpu": 3953.8,
|
||||
"tokens/trainable": 7607732
|
||||
},
|
||||
{
|
||||
"epoch": 0.35443037974683544,
|
||||
"grad_norm": 1.4296875,
|
||||
"learning_rate": 1.9597728560891266e-05,
|
||||
"loss": 1.1227096557617187,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 3.07317,
|
||||
"step": 70,
|
||||
"tokens/total": 9175040,
|
||||
"tokens/train_per_sec_per_gpu": 17981.63,
|
||||
"tokens/trainable": 8877038
|
||||
},
|
||||
{
|
||||
"epoch": 0.4050632911392405,
|
||||
"grad_norm": 1.3828125,
|
||||
"learning_rate": 1.9429759623974992e-05,
|
||||
"loss": 1.109241771697998,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 3.03206,
|
||||
"step": 80,
|
||||
"tokens/total": 10485760,
|
||||
"tokens/train_per_sec_per_gpu": 17934.18,
|
||||
"tokens/trainable": 10145302
|
||||
},
|
||||
{
|
||||
"epoch": 0.45569620253164556,
|
||||
"grad_norm": 1.2578125,
|
||||
"learning_rate": 1.9233550461078114e-05,
|
||||
"loss": 1.071034049987793,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.9184,
|
||||
"step": 90,
|
||||
"tokens/total": 11796480,
|
||||
"tokens/train_per_sec_per_gpu": 17952.48,
|
||||
"tokens/trainable": 11415878
|
||||
},
|
||||
{
|
||||
"epoch": 0.5063291139240507,
|
||||
"grad_norm": 1.2421875,
|
||||
"learning_rate": 1.900968867902419e-05,
|
||||
"loss": 1.0816166877746582,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.94944,
|
||||
"step": 100,
|
||||
"tokens/total": 13107200,
|
||||
"tokens/train_per_sec_per_gpu": 17943.02,
|
||||
"tokens/trainable": 12684573
|
||||
},
|
||||
{
|
||||
"epoch": 0.5063291139240507,
|
||||
"eval_indic_sft_mini_val_loss": 1.0623797178268433,
|
||||
"eval_indic_sft_mini_val_runtime": 157.7495,
|
||||
"eval_indic_sft_mini_val_samples_per_second": 12.171,
|
||||
"eval_indic_sft_mini_val_steps_per_second": 3.043,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 25.99,
|
||||
"memory/max_allocated (GiB)": 25.99,
|
||||
"step": 100
|
||||
},
|
||||
{
|
||||
"epoch": 0.5063291139240507,
|
||||
"eval_tulu_sft_mini_val_loss": 2.2967746257781982,
|
||||
"eval_tulu_sft_mini_val_runtime": 92.342,
|
||||
"eval_tulu_sft_mini_val_samples_per_second": 12.475,
|
||||
"eval_tulu_sft_mini_val_steps_per_second": 3.119,
|
||||
"memory/device_reserved (GiB)": 26.58,
|
||||
"memory/max_active (GiB)": 25.99,
|
||||
"memory/max_allocated (GiB)": 25.99,
|
||||
"step": 100
|
||||
},
|
||||
{
|
||||
"epoch": 0.5569620253164557,
|
||||
"grad_norm": 1.2421875,
|
||||
"learning_rate": 1.8758844698647457e-05,
|
||||
"loss": 1.106839370727539,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 3.02478,
|
||||
"step": 110,
|
||||
"tokens/total": 14417920,
|
||||
"tokens/train_per_sec_per_gpu": 3955.64,
|
||||
"tokens/trainable": 13951995
|
||||
},
|
||||
{
|
||||
"epoch": 0.6075949367088608,
|
||||
"grad_norm": 1.8984375,
|
||||
"learning_rate": 1.848176974701775e-05,
|
||||
"loss": 1.0786801338195802,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.9408,
|
||||
"step": 120,
|
||||
"tokens/total": 15728640,
|
||||
"tokens/train_per_sec_per_gpu": 17975.05,
|
||||
"tokens/trainable": 15221447
|
||||
},
|
||||
{
|
||||
"epoch": 0.6582278481012658,
|
||||
"grad_norm": 1.28125,
|
||||
"learning_rate": 1.8179293607667177e-05,
|
||||
"loss": 1.0739567756652832,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.92694,
|
||||
"step": 130,
|
||||
"tokens/total": 17039360,
|
||||
"tokens/train_per_sec_per_gpu": 17954.31,
|
||||
"tokens/trainable": 16490903
|
||||
},
|
||||
{
|
||||
"epoch": 0.7088607594936709,
|
||||
"grad_norm": 1.2890625,
|
||||
"learning_rate": 1.7852322135555946e-05,
|
||||
"loss": 1.0759190559387206,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.93269,
|
||||
"step": 140,
|
||||
"tokens/total": 18350080,
|
||||
"tokens/train_per_sec_per_gpu": 17887.36,
|
||||
"tokens/trainable": 17757184
|
||||
},
|
||||
{
|
||||
"epoch": 0.759493670886076,
|
||||
"grad_norm": 1.3828125,
|
||||
"learning_rate": 1.7501834544219697e-05,
|
||||
"loss": 1.0366504669189454,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.81976,
|
||||
"step": 150,
|
||||
"tokens/total": 19660800,
|
||||
"tokens/train_per_sec_per_gpu": 17911.77,
|
||||
"tokens/trainable": 19024962
|
||||
},
|
||||
{
|
||||
"epoch": 0.759493670886076,
|
||||
"eval_indic_sft_mini_val_loss": 1.0453351736068726,
|
||||
"eval_indic_sft_mini_val_runtime": 157.7623,
|
||||
"eval_indic_sft_mini_val_samples_per_second": 12.17,
|
||||
"eval_indic_sft_mini_val_steps_per_second": 3.043,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 25.99,
|
||||
"memory/max_allocated (GiB)": 25.99,
|
||||
"step": 150
|
||||
},
|
||||
{
|
||||
"epoch": 0.759493670886076,
|
||||
"eval_tulu_sft_mini_val_loss": 2.2883615493774414,
|
||||
"eval_tulu_sft_mini_val_runtime": 92.4616,
|
||||
"eval_tulu_sft_mini_val_samples_per_second": 12.459,
|
||||
"eval_tulu_sft_mini_val_steps_per_second": 3.115,
|
||||
"memory/device_reserved (GiB)": 26.58,
|
||||
"memory/max_active (GiB)": 25.99,
|
||||
"memory/max_allocated (GiB)": 25.99,
|
||||
"step": 150
|
||||
},
|
||||
{
|
||||
"epoch": 0.810126582278481,
|
||||
"grad_norm": 1.328125,
|
||||
"learning_rate": 1.7128880473222688e-05,
|
||||
"loss": 1.0812637329101562,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.9484,
|
||||
"step": 160,
|
||||
"tokens/total": 20971520,
|
||||
"tokens/train_per_sec_per_gpu": 3951.59,
|
||||
"tokens/trainable": 20291324
|
||||
},
|
||||
{
|
||||
"epoch": 0.8607594936708861,
|
||||
"grad_norm": 1.328125,
|
||||
"learning_rate": 1.6734576844699234e-05,
|
||||
"loss": 1.0667606353759767,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.90595,
|
||||
"step": 170,
|
||||
"tokens/total": 22282240,
|
||||
"tokens/train_per_sec_per_gpu": 17990.97,
|
||||
"tokens/trainable": 21559984
|
||||
},
|
||||
{
|
||||
"epoch": 0.9113924050632911,
|
||||
"grad_norm": 1.3515625,
|
||||
"learning_rate": 1.6320104518397473e-05,
|
||||
"loss": 1.067934513092041,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.90936,
|
||||
"step": 180,
|
||||
"tokens/total": 23592960,
|
||||
"tokens/train_per_sec_per_gpu": 17944.73,
|
||||
"tokens/trainable": 22828372
|
||||
},
|
||||
{
|
||||
"epoch": 0.9620253164556962,
|
||||
"grad_norm": 1.296875,
|
||||
"learning_rate": 1.588670475524283e-05,
|
||||
"loss": 1.04850492477417,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.85338,
|
||||
"step": 190,
|
||||
"tokens/total": 24903680,
|
||||
"tokens/train_per_sec_per_gpu": 17890.69,
|
||||
"tokens/trainable": 24094636
|
||||
},
|
||||
{
|
||||
"epoch": 1.010126582278481,
|
||||
"grad_norm": 1.3046875,
|
||||
"learning_rate": 1.5435675500012212e-05,
|
||||
"loss": 1.0491924285888672,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.85534,
|
||||
"step": 200,
|
||||
"tokens/total": 26136576,
|
||||
"tokens/train_per_sec_per_gpu": 17465.18,
|
||||
"tokens/trainable": 25286404
|
||||
},
|
||||
{
|
||||
"epoch": 1.010126582278481,
|
||||
"eval_indic_sft_mini_val_loss": 1.0354256629943848,
|
||||
"eval_indic_sft_mini_val_runtime": 157.741,
|
||||
"eval_indic_sft_mini_val_samples_per_second": 12.172,
|
||||
"eval_indic_sft_mini_val_steps_per_second": 3.043,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 25.99,
|
||||
"memory/max_allocated (GiB)": 25.99,
|
||||
"step": 200
|
||||
},
|
||||
{
|
||||
"epoch": 1.010126582278481,
|
||||
"eval_tulu_sft_mini_val_loss": 2.291062593460083,
|
||||
"eval_tulu_sft_mini_val_runtime": 92.344,
|
||||
"eval_tulu_sft_mini_val_samples_per_second": 12.475,
|
||||
"eval_tulu_sft_mini_val_steps_per_second": 3.119,
|
||||
"memory/device_reserved (GiB)": 26.58,
|
||||
"memory/max_active (GiB)": 25.99,
|
||||
"memory/max_allocated (GiB)": 25.99,
|
||||
"step": 200
|
||||
},
|
||||
{
|
||||
"epoch": 1.0607594936708862,
|
||||
"grad_norm": 1.375,
|
||||
"learning_rate": 1.4968367494251486e-05,
|
||||
"loss": 1.0487144470214844,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.85398,
|
||||
"step": 210,
|
||||
"tokens/total": 27447296,
|
||||
"tokens/train_per_sec_per_gpu": 3957.26,
|
||||
"tokens/trainable": 26554084
|
||||
},
|
||||
{
|
||||
"epoch": 1.111392405063291,
|
||||
"grad_norm": 1.3671875,
|
||||
"learning_rate": 1.4486180231077278e-05,
|
||||
"loss": 1.0355692863464356,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.81671,
|
||||
"step": 220,
|
||||
"tokens/total": 28758016,
|
||||
"tokens/train_per_sec_per_gpu": 17979.8,
|
||||
"tokens/trainable": 27823280
|
||||
},
|
||||
{
|
||||
"epoch": 1.1620253164556962,
|
||||
"grad_norm": 1.265625,
|
||||
"learning_rate": 1.3990557763977694e-05,
|
||||
"loss": 1.0380287170410156,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.82365,
|
||||
"step": 230,
|
||||
"tokens/total": 30068736,
|
||||
"tokens/train_per_sec_per_gpu": 17964.75,
|
||||
"tokens/trainable": 29092706
|
||||
},
|
||||
{
|
||||
"epoch": 1.2126582278481013,
|
||||
"grad_norm": 1.2734375,
|
||||
"learning_rate": 1.3482984382163713e-05,
|
||||
"loss": 1.036821174621582,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.82024,
|
||||
"step": 240,
|
||||
"tokens/total": 31379456,
|
||||
"tokens/train_per_sec_per_gpu": 17930.26,
|
||||
"tokens/trainable": 30360732
|
||||
},
|
||||
{
|
||||
"epoch": 1.2632911392405064,
|
||||
"grad_norm": 1.2265625,
|
||||
"learning_rate": 1.2964980165422701e-05,
|
||||
"loss": 0.995778751373291,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.70683,
|
||||
"step": 250,
|
||||
"tokens/total": 32690176,
|
||||
"tokens/train_per_sec_per_gpu": 17927.94,
|
||||
"tokens/trainable": 31631352
|
||||
},
|
||||
{
|
||||
"epoch": 1.2632911392405064,
|
||||
"eval_indic_sft_mini_val_loss": 1.027944803237915,
|
||||
"eval_indic_sft_mini_val_runtime": 157.8775,
|
||||
"eval_indic_sft_mini_val_samples_per_second": 12.161,
|
||||
"eval_indic_sft_mini_val_steps_per_second": 3.04,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 25.99,
|
||||
"memory/max_allocated (GiB)": 25.99,
|
||||
"step": 250
|
||||
},
|
||||
{
|
||||
"epoch": 1.2632911392405064,
|
||||
"eval_tulu_sft_mini_val_loss": 2.2984113693237305,
|
||||
"eval_tulu_sft_mini_val_runtime": 92.3048,
|
||||
"eval_tulu_sft_mini_val_samples_per_second": 12.48,
|
||||
"eval_tulu_sft_mini_val_steps_per_second": 3.12,
|
||||
"memory/device_reserved (GiB)": 26.58,
|
||||
"memory/max_active (GiB)": 25.99,
|
||||
"memory/max_allocated (GiB)": 25.99,
|
||||
"step": 250
|
||||
},
|
||||
{
|
||||
"epoch": 1.3139240506329113,
|
||||
"grad_norm": 1.2109375,
|
||||
"learning_rate": 1.2438096431786408e-05,
|
||||
"loss": 1.0438777923583984,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.84021,
|
||||
"step": 260,
|
||||
"tokens/total": 34000896,
|
||||
"tokens/train_per_sec_per_gpu": 3955.89,
|
||||
"tokens/trainable": 32899468
|
||||
},
|
||||
{
|
||||
"epoch": 1.3645569620253164,
|
||||
"grad_norm": 1.28125,
|
||||
"learning_rate": 1.1903911091646684e-05,
|
||||
"loss": 1.0240283012390137,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.78439,
|
||||
"step": 270,
|
||||
"tokens/total": 35311616,
|
||||
"tokens/train_per_sec_per_gpu": 17961.08,
|
||||
"tokens/trainable": 34168384
|
||||
},
|
||||
{
|
||||
"epoch": 1.4151898734177215,
|
||||
"grad_norm": 1.234375,
|
||||
"learning_rate": 1.1364023922232503e-05,
|
||||
"loss": 1.031261920928955,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.8046,
|
||||
"step": 280,
|
||||
"tokens/total": 36622336,
|
||||
"tokens/train_per_sec_per_gpu": 17939.56,
|
||||
"tokens/trainable": 35436704
|
||||
},
|
||||
{
|
||||
"epoch": 1.4658227848101266,
|
||||
"grad_norm": 1.3359375,
|
||||
"learning_rate": 1.0820051776600175e-05,
|
||||
"loss": 1.0369884490966796,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.82071,
|
||||
"step": 290,
|
||||
"tokens/total": 37933056,
|
||||
"tokens/train_per_sec_per_gpu": 17936.98,
|
||||
"tokens/trainable": 36705368
|
||||
},
|
||||
{
|
||||
"epoch": 1.5164556962025317,
|
||||
"grad_norm": 1.1875,
|
||||
"learning_rate": 1.0273623741484924e-05,
|
||||
"loss": 1.0238310813903808,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.78384,
|
||||
"step": 300,
|
||||
"tokens/total": 39243776,
|
||||
"tokens/train_per_sec_per_gpu": 17926.64,
|
||||
"tokens/trainable": 37973400
|
||||
},
|
||||
{
|
||||
"epoch": 1.5164556962025317,
|
||||
"eval_indic_sft_mini_val_loss": 1.0224663019180298,
|
||||
"eval_indic_sft_mini_val_runtime": 157.771,
|
||||
"eval_indic_sft_mini_val_samples_per_second": 12.17,
|
||||
"eval_indic_sft_mini_val_steps_per_second": 3.042,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 25.99,
|
||||
"memory/max_allocated (GiB)": 25.99,
|
||||
"step": 300
|
||||
},
|
||||
{
|
||||
"epoch": 1.5164556962025317,
|
||||
"eval_tulu_sft_mini_val_loss": 2.295070171356201,
|
||||
"eval_tulu_sft_mini_val_runtime": 92.349,
|
||||
"eval_tulu_sft_mini_val_samples_per_second": 12.474,
|
||||
"eval_tulu_sft_mini_val_steps_per_second": 3.119,
|
||||
"memory/device_reserved (GiB)": 26.58,
|
||||
"memory/max_active (GiB)": 25.99,
|
||||
"memory/max_allocated (GiB)": 25.99,
|
||||
"step": 300
|
||||
},
|
||||
{
|
||||
"epoch": 1.5670886075949366,
|
||||
"grad_norm": 1.2421875,
|
||||
"learning_rate": 9.726376258515077e-06,
|
||||
"loss": 1.0644823074340821,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.89934,
|
||||
"step": 310,
|
||||
"tokens/total": 40554496,
|
||||
"tokens/train_per_sec_per_gpu": 3956.31,
|
||||
"tokens/trainable": 39240980
|
||||
},
|
||||
{
|
||||
"epoch": 1.6177215189873417,
|
||||
"grad_norm": 1.234375,
|
||||
"learning_rate": 9.179948223399828e-06,
|
||||
"loss": 1.0305088996887206,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.80249,
|
||||
"step": 320,
|
||||
"tokens/total": 41865216,
|
||||
"tokens/train_per_sec_per_gpu": 17965.06,
|
||||
"tokens/trainable": 40509632
|
||||
},
|
||||
{
|
||||
"epoch": 1.6683544303797468,
|
||||
"grad_norm": 1.265625,
|
||||
"learning_rate": 8.6359760777675e-06,
|
||||
"loss": 1.0250298500061035,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.78718,
|
||||
"step": 330,
|
||||
"tokens/total": 43175936,
|
||||
"tokens/train_per_sec_per_gpu": 17953.88,
|
||||
"tokens/trainable": 41777768
|
||||
},
|
||||
{
|
||||
"epoch": 1.7189873417721517,
|
||||
"grad_norm": 1.2265625,
|
||||
"learning_rate": 8.096088908353316e-06,
|
||||
"loss": 1.0106207847595214,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.74731,
|
||||
"step": 340,
|
||||
"tokens/total": 44486656,
|
||||
"tokens/train_per_sec_per_gpu": 17933.81,
|
||||
"tokens/trainable": 43045456
|
||||
},
|
||||
{
|
||||
"epoch": 1.769620253164557,
|
||||
"grad_norm": 1.2578125,
|
||||
"learning_rate": 7.561903568213595e-06,
|
||||
"loss": 0.9956131935119629,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.70638,
|
||||
"step": 350,
|
||||
"tokens/total": 45797376,
|
||||
"tokens/train_per_sec_per_gpu": 17927.21,
|
||||
"tokens/trainable": 44311528
|
||||
},
|
||||
{
|
||||
"epoch": 1.769620253164557,
|
||||
"eval_indic_sft_mini_val_loss": 1.0202709436416626,
|
||||
"eval_indic_sft_mini_val_runtime": 157.7597,
|
||||
"eval_indic_sft_mini_val_samples_per_second": 12.17,
|
||||
"eval_indic_sft_mini_val_steps_per_second": 3.043,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 25.99,
|
||||
"memory/max_allocated (GiB)": 25.99,
|
||||
"step": 350
|
||||
},
|
||||
{
|
||||
"epoch": 1.769620253164557,
|
||||
"eval_tulu_sft_mini_val_loss": 2.2989728450775146,
|
||||
"eval_tulu_sft_mini_val_runtime": 92.4784,
|
||||
"eval_tulu_sft_mini_val_samples_per_second": 12.457,
|
||||
"eval_tulu_sft_mini_val_steps_per_second": 3.114,
|
||||
"memory/device_reserved (GiB)": 26.58,
|
||||
"memory/max_active (GiB)": 25.99,
|
||||
"memory/max_allocated (GiB)": 25.99,
|
||||
"step": 350
|
||||
},
|
||||
{
|
||||
"epoch": 1.820253164556962,
|
||||
"grad_norm": 1.2734375,
|
||||
"learning_rate": 7.035019834577301e-06,
|
||||
"loss": 1.0437658309936524,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.83989,
|
||||
"step": 360,
|
||||
"tokens/total": 47108096,
|
||||
"tokens/train_per_sec_per_gpu": 3952.58,
|
||||
"tokens/trainable": 45578288
|
||||
},
|
||||
{
|
||||
"epoch": 1.870886075949367,
|
||||
"grad_norm": 1.2265625,
|
||||
"learning_rate": 6.517015617836292e-06,
|
||||
"loss": 1.0040513038635255,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.72932,
|
||||
"step": 370,
|
||||
"tokens/total": 48418816,
|
||||
"tokens/train_per_sec_per_gpu": 17939.3,
|
||||
"tokens/trainable": 46846324
|
||||
},
|
||||
{
|
||||
"epoch": 1.9215189873417722,
|
||||
"grad_norm": 1.234375,
|
||||
"learning_rate": 6.009442236022307e-06,
|
||||
"loss": 1.0151902198791505,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.75989,
|
||||
"step": 380,
|
||||
"tokens/total": 49729536,
|
||||
"tokens/train_per_sec_per_gpu": 17915.04,
|
||||
"tokens/trainable": 48113428
|
||||
},
|
||||
{
|
||||
"epoch": 1.972151898734177,
|
||||
"grad_norm": 1.3125,
|
||||
"learning_rate": 5.513819768922723e-06,
|
||||
"loss": 0.997506046295166,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.71151,
|
||||
"step": 390,
|
||||
"tokens/total": 51040256,
|
||||
"tokens/train_per_sec_per_gpu": 17963.73,
|
||||
"tokens/trainable": 49382776
|
||||
},
|
||||
{
|
||||
"epoch": 2.020253164556962,
|
||||
"grad_norm": 1.2421875,
|
||||
"learning_rate": 5.031632505748516e-06,
|
||||
"loss": 1.0011167526245117,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.72132,
|
||||
"step": 400,
|
||||
"tokens/total": 52273152,
|
||||
"tokens/train_per_sec_per_gpu": 17458.66,
|
||||
"tokens/trainable": 50572704
|
||||
},
|
||||
{
|
||||
"epoch": 2.020253164556962,
|
||||
"eval_indic_sft_mini_val_loss": 1.0187087059020996,
|
||||
"eval_indic_sft_mini_val_runtime": 158.0242,
|
||||
"eval_indic_sft_mini_val_samples_per_second": 12.15,
|
||||
"eval_indic_sft_mini_val_steps_per_second": 3.038,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 25.99,
|
||||
"memory/max_allocated (GiB)": 25.99,
|
||||
"step": 400
|
||||
},
|
||||
{
|
||||
"epoch": 2.020253164556962,
|
||||
"eval_tulu_sft_mini_val_loss": 2.2966930866241455,
|
||||
"eval_tulu_sft_mini_val_runtime": 92.3222,
|
||||
"eval_tulu_sft_mini_val_samples_per_second": 12.478,
|
||||
"eval_tulu_sft_mini_val_steps_per_second": 3.12,
|
||||
"memory/device_reserved (GiB)": 26.58,
|
||||
"memory/max_active (GiB)": 25.99,
|
||||
"memory/max_allocated (GiB)": 25.99,
|
||||
"step": 400
|
||||
},
|
||||
{
|
||||
"epoch": 2.070886075949367,
|
||||
"grad_norm": 1.2421875,
|
||||
"learning_rate": 4.56432449998779e-06,
|
||||
"loss": 1.0103323936462403,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.74651,
|
||||
"step": 410,
|
||||
"tokens/total": 53583872,
|
||||
"tokens/train_per_sec_per_gpu": 3952.63,
|
||||
"tokens/trainable": 51839796
|
||||
},
|
||||
{
|
||||
"epoch": 2.1215189873417724,
|
||||
"grad_norm": 1.234375,
|
||||
"learning_rate": 4.113295244757171e-06,
|
||||
"loss": 1.0133058547973632,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.75469,
|
||||
"step": 420,
|
||||
"tokens/total": 54894592,
|
||||
"tokens/train_per_sec_per_gpu": 17984.79,
|
||||
"tokens/trainable": 53108816
|
||||
},
|
||||
{
|
||||
"epoch": 2.1721518987341772,
|
||||
"grad_norm": 1.21875,
|
||||
"learning_rate": 3.679895481602529e-06,
|
||||
"loss": 1.0214984893798829,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.77735,
|
||||
"step": 430,
|
||||
"tokens/total": 56205312,
|
||||
"tokens/train_per_sec_per_gpu": 17959.57,
|
||||
"tokens/trainable": 54378168
|
||||
},
|
||||
{
|
||||
"epoch": 2.222784810126582,
|
||||
"grad_norm": 1.2109375,
|
||||
"learning_rate": 3.2654231553007665e-06,
|
||||
"loss": 0.9840593338012695,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.67529,
|
||||
"step": 440,
|
||||
"tokens/total": 57516032,
|
||||
"tokens/train_per_sec_per_gpu": 17930.45,
|
||||
"tokens/trainable": 55645416
|
||||
},
|
||||
{
|
||||
"epoch": 2.2734177215189875,
|
||||
"grad_norm": 1.1953125,
|
||||
"learning_rate": 2.871119526777315e-06,
|
||||
"loss": 1.028823184967041,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.79777,
|
||||
"step": 450,
|
||||
"tokens/total": 58826752,
|
||||
"tokens/train_per_sec_per_gpu": 17944.17,
|
||||
"tokens/trainable": 56914256
|
||||
},
|
||||
{
|
||||
"epoch": 2.2734177215189875,
|
||||
"eval_indic_sft_mini_val_loss": 1.0182074308395386,
|
||||
"eval_indic_sft_mini_val_runtime": 157.9039,
|
||||
"eval_indic_sft_mini_val_samples_per_second": 12.159,
|
||||
"eval_indic_sft_mini_val_steps_per_second": 3.04,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 25.99,
|
||||
"memory/max_allocated (GiB)": 25.99,
|
||||
"step": 450
|
||||
},
|
||||
{
|
||||
"epoch": 2.2734177215189875,
|
||||
"eval_tulu_sft_mini_val_loss": 2.297419309616089,
|
||||
"eval_tulu_sft_mini_val_runtime": 92.4484,
|
||||
"eval_tulu_sft_mini_val_samples_per_second": 12.461,
|
||||
"eval_tulu_sft_mini_val_steps_per_second": 3.115,
|
||||
"memory/device_reserved (GiB)": 26.59,
|
||||
"memory/max_active (GiB)": 25.99,
|
||||
"memory/max_allocated (GiB)": 25.99,
|
||||
"step": 450
|
||||
},
|
||||
{
|
||||
"epoch": 2.3240506329113924,
|
||||
"grad_norm": 1.21875,
|
||||
"learning_rate": 2.4981654557803026e-06,
|
||||
"loss": 1.0112363815307617,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.749,
|
||||
"step": 460,
|
||||
"tokens/total": 60137472,
|
||||
"tokens/train_per_sec_per_gpu": 3954.28,
|
||||
"tokens/trainable": 58182604
|
||||
},
|
||||
{
|
||||
"epoch": 2.3746835443037977,
|
||||
"grad_norm": 1.265625,
|
||||
"learning_rate": 2.1476778644440553e-06,
|
||||
"loss": 0.9997093200683593,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.71749,
|
||||
"step": 470,
|
||||
"tokens/total": 61448192,
|
||||
"tokens/train_per_sec_per_gpu": 17956.34,
|
||||
"tokens/trainable": 59450748
|
||||
},
|
||||
{
|
||||
"epoch": 2.4253164556962026,
|
||||
"grad_norm": 1.21875,
|
||||
"learning_rate": 1.820706392332824e-06,
|
||||
"loss": 1.0180435180664062,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.76777,
|
||||
"step": 480,
|
||||
"tokens/total": 62758912,
|
||||
"tokens/train_per_sec_per_gpu": 17939.84,
|
||||
"tokens/trainable": 60720528
|
||||
},
|
||||
{
|
||||
"epoch": 2.4759493670886075,
|
||||
"grad_norm": 1.234375,
|
||||
"learning_rate": 1.518230252982248e-06,
|
||||
"loss": 1.040645217895508,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.83104,
|
||||
"step": 490,
|
||||
"tokens/total": 64069632,
|
||||
"tokens/train_per_sec_per_gpu": 17931.8,
|
||||
"tokens/trainable": 61988728
|
||||
},
|
||||
{
|
||||
"epoch": 2.526582278481013,
|
||||
"grad_norm": 1.171875,
|
||||
"learning_rate": 1.2411553013525457e-06,
|
||||
"loss": 1.0239150047302246,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 32.29,
|
||||
"memory/max_allocated (GiB)": 32.29,
|
||||
"ppl": 2.78407,
|
||||
"step": 500,
|
||||
"tokens/total": 65380352,
|
||||
"tokens/train_per_sec_per_gpu": 17916.35,
|
||||
"tokens/trainable": 63254896
|
||||
},
|
||||
{
|
||||
"epoch": 2.526582278481013,
|
||||
"eval_indic_sft_mini_val_loss": 1.018338680267334,
|
||||
"eval_indic_sft_mini_val_runtime": 157.6995,
|
||||
"eval_indic_sft_mini_val_samples_per_second": 12.175,
|
||||
"eval_indic_sft_mini_val_steps_per_second": 3.044,
|
||||
"memory/device_reserved (GiB)": 37.04,
|
||||
"memory/max_active (GiB)": 25.99,
|
||||
"memory/max_allocated (GiB)": 25.99,
|
||||
"step": 500
|
||||
},
|
||||
{
|
||||
"epoch": 2.526582278481013,
|
||||
"eval_tulu_sft_mini_val_loss": 2.299147605895996,
|
||||
"eval_tulu_sft_mini_val_runtime": 92.1488,
|
||||
"eval_tulu_sft_mini_val_samples_per_second": 12.502,
|
||||
"eval_tulu_sft_mini_val_steps_per_second": 3.125,
|
||||
"memory/device_reserved (GiB)": 26.59,
|
||||
"memory/max_active (GiB)": 25.99,
|
||||
"memory/max_allocated (GiB)": 25.99,
|
||||
"step": 500
|
||||
}
|
||||
],
|
||||
"logging_steps": 10,
|
||||
"max_steps": 591,
|
||||
"num_input_tokens_seen": 0,
|
||||
"num_train_epochs": 3,
|
||||
"save_steps": 500,
|
||||
"stateful_callbacks": {
|
||||
"TrainerControl": {
|
||||
"args": {
|
||||
"should_epoch_stop": false,
|
||||
"should_evaluate": false,
|
||||
"should_log": false,
|
||||
"should_save": true,
|
||||
"should_training_stop": false
|
||||
},
|
||||
"attributes": {}
|
||||
}
|
||||
},
|
||||
"total_flos": 1.4039702725617254e+17,
|
||||
"train_batch_size": 4,
|
||||
"trial_name": null,
|
||||
"trial_params": null
|
||||
}
|
||||
3
checkpoint-500/training_args.bin
Normal file
3
checkpoint-500/training_args.bin
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:f1ab83fb7a53e79e01752f8af988eed016a56cd8159a16af0f41fcfaaa076803
|
||||
size 9169
|
||||
54
checkpoint-591/chat_template.jinja
Normal file
54
checkpoint-591/chat_template.jinja
Normal file
@@ -0,0 +1,54 @@
|
||||
{%- if tools %}
|
||||
{{- '<|im_start|>system\n' }}
|
||||
{%- if messages[0]['role'] == 'system' %}
|
||||
{{- messages[0]['content'] }}
|
||||
{%- else %}
|
||||
{{- 'You are a helpful assistant.' }}
|
||||
{%- endif %}
|
||||
{{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
|
||||
{%- for tool in tools %}
|
||||
{{- "\n" }}
|
||||
{{- tool | tojson }}
|
||||
{%- endfor %}
|
||||
{{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
|
||||
{%- else %}
|
||||
{%- if messages[0]['role'] == 'system' %}
|
||||
{{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
|
||||
{%- else %}
|
||||
{{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
|
||||
{%- endif %}
|
||||
{%- endif %}
|
||||
{%- for message in messages %}
|
||||
{%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
|
||||
{{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
|
||||
{%- elif message.role == "assistant" %}
|
||||
{{- '<|im_start|>' + message.role }}
|
||||
{%- if message.content %}
|
||||
{{- '\n' + message.content }}
|
||||
{%- endif %}
|
||||
{%- for tool_call in message.tool_calls %}
|
||||
{%- if tool_call.function is defined %}
|
||||
{%- set tool_call = tool_call.function %}
|
||||
{%- endif %}
|
||||
{{- '\n<tool_call>\n{"name": "' }}
|
||||
{{- tool_call.name }}
|
||||
{{- '", "arguments": ' }}
|
||||
{{- tool_call.arguments | tojson }}
|
||||
{{- '}\n</tool_call>' }}
|
||||
{%- endfor %}
|
||||
{{- '<|im_end|>\n' }}
|
||||
{%- elif message.role == "tool" %}
|
||||
{%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
|
||||
{{- '<|im_start|>user' }}
|
||||
{%- endif %}
|
||||
{{- '\n<tool_response>\n' }}
|
||||
{{- message.content }}
|
||||
{{- '\n</tool_response>' }}
|
||||
{%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
|
||||
{{- '<|im_end|>\n' }}
|
||||
{%- endif %}
|
||||
{%- endif %}
|
||||
{%- endfor %}
|
||||
{%- if add_generation_prompt %}
|
||||
{{- '<|im_start|>assistant\n' }}
|
||||
{%- endif %}
|
||||
58
checkpoint-591/config.json
Normal file
58
checkpoint-591/config.json
Normal file
@@ -0,0 +1,58 @@
|
||||
{
|
||||
"architectures": [
|
||||
"Qwen2ForCausalLM"
|
||||
],
|
||||
"attention_dropout": 0.0,
|
||||
"bos_token_id": null,
|
||||
"dtype": "bfloat16",
|
||||
"eos_token_id": 151643,
|
||||
"hidden_act": "silu",
|
||||
"hidden_size": 896,
|
||||
"initializer_range": 0.02,
|
||||
"intermediate_size": 4864,
|
||||
"layer_types": [
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention"
|
||||
],
|
||||
"max_position_embeddings": 32768,
|
||||
"max_window_layers": 24,
|
||||
"model_type": "qwen2",
|
||||
"num_attention_heads": 14,
|
||||
"num_hidden_layers": 24,
|
||||
"num_key_value_heads": 2,
|
||||
"pad_token_id": 151643,
|
||||
"rms_norm_eps": 1e-06,
|
||||
"rope_parameters": {
|
||||
"rope_theta": 1000000.0,
|
||||
"rope_type": "default"
|
||||
},
|
||||
"sliding_window": null,
|
||||
"tie_word_embeddings": true,
|
||||
"transformers_version": "5.14.1",
|
||||
"use_cache": false,
|
||||
"use_mrope": false,
|
||||
"use_sliding_window": false,
|
||||
"vocab_size": 151936
|
||||
}
|
||||
9
checkpoint-591/generation_config.json
Normal file
9
checkpoint-591/generation_config.json
Normal file
@@ -0,0 +1,9 @@
|
||||
{
|
||||
"do_sample": true,
|
||||
"eos_token_id": [
|
||||
151643
|
||||
],
|
||||
"max_new_tokens": 2048,
|
||||
"pad_token_id": 151643,
|
||||
"transformers_version": "5.14.1"
|
||||
}
|
||||
3
checkpoint-591/model.safetensors
Normal file
3
checkpoint-591/model.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:4f2ef297270ca28cd4a413f624c81a6c0e639cc6667fbbfd6fdab6640a114eb6
|
||||
size 988097824
|
||||
3
checkpoint-591/tokenizer.json
Normal file
3
checkpoint-591/tokenizer.json
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:3fd169731d2cbde95e10bf356d66d5997fd885dd8dbb6fb4684da3f23b2585d8
|
||||
size 11421892
|
||||
30
checkpoint-591/tokenizer_config.json
Normal file
30
checkpoint-591/tokenizer_config.json
Normal file
@@ -0,0 +1,30 @@
|
||||
{
|
||||
"add_prefix_space": false,
|
||||
"backend": "tokenizers",
|
||||
"bos_token": null,
|
||||
"clean_up_tokenization_spaces": false,
|
||||
"eos_token": "<|endoftext|>",
|
||||
"errors": "replace",
|
||||
"extra_special_tokens": [
|
||||
"<|im_start|>",
|
||||
"<|im_end|>",
|
||||
"<|object_ref_start|>",
|
||||
"<|object_ref_end|>",
|
||||
"<|box_start|>",
|
||||
"<|box_end|>",
|
||||
"<|quad_start|>",
|
||||
"<|quad_end|>",
|
||||
"<|vision_start|>",
|
||||
"<|vision_end|>",
|
||||
"<|vision_pad|>",
|
||||
"<|image_pad|>",
|
||||
"<|video_pad|>"
|
||||
],
|
||||
"is_local": false,
|
||||
"local_files_only": false,
|
||||
"model_max_length": 131072,
|
||||
"pad_token": "<|endoftext|>",
|
||||
"split_special_tokens": false,
|
||||
"tokenizer_class": "Qwen2Tokenizer",
|
||||
"unk_token": null
|
||||
}
|
||||
1
checkpoint-591/tokens_state.json
Normal file
1
checkpoint-591/tokens_state.json
Normal file
@@ -0,0 +1 @@
|
||||
{"total": 77307904, "trainable": 74795984}
|
||||
1145
checkpoint-591/trainer_state.json
Normal file
1145
checkpoint-591/trainer_state.json
Normal file
File diff suppressed because it is too large
Load Diff
3
checkpoint-591/training_args.bin
Normal file
3
checkpoint-591/training_args.bin
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:f1ab83fb7a53e79e01752f8af988eed016a56cd8159a16af0f41fcfaaa076803
|
||||
size 9169
|
||||
58
config.json
Normal file
58
config.json
Normal file
@@ -0,0 +1,58 @@
|
||||
{
|
||||
"architectures": [
|
||||
"Qwen2ForCausalLM"
|
||||
],
|
||||
"attention_dropout": 0.0,
|
||||
"bos_token_id": null,
|
||||
"dtype": "bfloat16",
|
||||
"eos_token_id": 151643,
|
||||
"hidden_act": "silu",
|
||||
"hidden_size": 896,
|
||||
"initializer_range": 0.02,
|
||||
"intermediate_size": 4864,
|
||||
"layer_types": [
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention"
|
||||
],
|
||||
"max_position_embeddings": 32768,
|
||||
"max_window_layers": 24,
|
||||
"model_type": "qwen2",
|
||||
"num_attention_heads": 14,
|
||||
"num_hidden_layers": 24,
|
||||
"num_key_value_heads": 2,
|
||||
"pad_token_id": 151643,
|
||||
"rms_norm_eps": 1e-06,
|
||||
"rope_parameters": {
|
||||
"rope_theta": 1000000.0,
|
||||
"rope_type": "default"
|
||||
},
|
||||
"sliding_window": null,
|
||||
"tie_word_embeddings": true,
|
||||
"transformers_version": "5.14.1",
|
||||
"use_cache": false,
|
||||
"use_mrope": false,
|
||||
"use_sliding_window": false,
|
||||
"vocab_size": 151936
|
||||
}
|
||||
9
generation_config.json
Normal file
9
generation_config.json
Normal file
@@ -0,0 +1,9 @@
|
||||
{
|
||||
"do_sample": true,
|
||||
"eos_token_id": [
|
||||
151643
|
||||
],
|
||||
"max_new_tokens": 2048,
|
||||
"pad_token_id": 151643,
|
||||
"transformers_version": "5.14.1"
|
||||
}
|
||||
3
model.safetensors
Normal file
3
model.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:4f2ef297270ca28cd4a413f624c81a6c0e639cc6667fbbfd6fdab6640a114eb6
|
||||
size 988097824
|
||||
3
tokenizer.json
Normal file
3
tokenizer.json
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:3fd169731d2cbde95e10bf356d66d5997fd885dd8dbb6fb4684da3f23b2585d8
|
||||
size 11421892
|
||||
30
tokenizer_config.json
Normal file
30
tokenizer_config.json
Normal file
@@ -0,0 +1,30 @@
|
||||
{
|
||||
"add_prefix_space": false,
|
||||
"backend": "tokenizers",
|
||||
"bos_token": null,
|
||||
"clean_up_tokenization_spaces": false,
|
||||
"eos_token": "<|endoftext|>",
|
||||
"errors": "replace",
|
||||
"extra_special_tokens": [
|
||||
"<|im_start|>",
|
||||
"<|im_end|>",
|
||||
"<|object_ref_start|>",
|
||||
"<|object_ref_end|>",
|
||||
"<|box_start|>",
|
||||
"<|box_end|>",
|
||||
"<|quad_start|>",
|
||||
"<|quad_end|>",
|
||||
"<|vision_start|>",
|
||||
"<|vision_end|>",
|
||||
"<|vision_pad|>",
|
||||
"<|image_pad|>",
|
||||
"<|video_pad|>"
|
||||
],
|
||||
"is_local": false,
|
||||
"local_files_only": false,
|
||||
"model_max_length": 131072,
|
||||
"pad_token": "<|endoftext|>",
|
||||
"split_special_tokens": false,
|
||||
"tokenizer_class": "Qwen2Tokenizer",
|
||||
"unk_token": null
|
||||
}
|
||||
3
training_args.bin
Normal file
3
training_args.bin
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:f1ab83fb7a53e79e01752f8af988eed016a56cd8159a16af0f41fcfaaa076803
|
||||
size 9169
|
||||
Reference in New Issue
Block a user