初始化项目,由ModelHub XC社区提供模型

Model: robertspumiaca1975/Qwen2.5-Coder-14B-n8n-Workflow-Generator
Source: Original Platform
This commit is contained in:
ModelHub XC
2026-09-12 18:46:20 +08:00
commit fca4cc8081
46 changed files with 458160 additions and 0 deletions

613
debug.log Normal file
View File

@@ -0,0 +1,613 @@
[2025-12-25 22:27:59,815] [DEBUG] [axolotl.utils.config.log_gpu_memory_usage:127] [PID:1133] baseline 0.000GB ()
[2025-12-25 22:27:59,815] [INFO] [axolotl.cli.config.load_cfg:248] [PID:1133] config:
{
"activation_offloading": false,
"adapter": "qlora",
"axolotl_config_path": "config.yaml",
"base_model": "Qwen/Qwen2.5-Coder-14B-Instruct",
"base_model_config": "Qwen/Qwen2.5-Coder-14B-Instruct",
"batch_size": 16,
"bf16": true,
"capabilities": {
"bf16": true,
"compute_capability": "sm_90",
"fp8": false,
"n_gpu": 1,
"n_node": 1
},
"context_parallel_size": 1,
"dataloader_num_workers": 1,
"dataloader_pin_memory": true,
"dataloader_prefetch_factor": 256,
"dataset_processes": 36,
"datasets": [
{
"message_property_mappings": {
"content": "content",
"role": "role"
},
"path": "mbakgun/n8nbuilder-n8n-workflows-dataset",
"trust_remote_code": false,
"type": "alpaca"
}
],
"ddp": false,
"device": "cuda:0",
"dion_rank_fraction": 1.0,
"dion_rank_multiple_of": 1,
"env_capabilities": {
"torch_version": "2.7.1"
},
"eval_batch_size": 1,
"eval_causal_lm_metrics": [
"sacrebleu",
"comet",
"ter",
"chrf"
],
"eval_max_new_tokens": 128,
"eval_table_size": 0,
"experimental_skip_move_to_device": true,
"flash_attention": true,
"fp16": false,
"gradient_accumulation_steps": 16,
"gradient_checkpointing": true,
"gradient_checkpointing_kwargs": {
"use_reentrant": false
},
"include_tkps": true,
"learning_rate": 0.0002,
"lisa_layers_attribute": "model.layers",
"load_best_model_at_end": false,
"load_in_4bit": true,
"load_in_8bit": false,
"local_rank": 0,
"logging_steps": 1,
"lora_alpha": 64,
"lora_dropout": 0.05,
"lora_r": 32,
"lora_target_modules": [
"q_proj",
"k_proj",
"v_proj",
"o_proj",
"gate_proj",
"up_proj",
"down_proj"
],
"loraplus_lr_embedding": 1e-06,
"lr_scheduler": "cosine",
"mean_resizing_embeddings": false,
"micro_batch_size": 1,
"model_config_type": "qwen2",
"num_epochs": 3.0,
"optimizer": "adamw_bnb_8bit",
"output_dir": "./outputs/qwen25-coder-n8n",
"pad_to_sequence_len": false,
"pretrain_multipack_attn": true,
"profiler_steps_start": 0,
"qlora_sharded_model_loading": false,
"ray_num_workers": 1,
"resources_per_worker": {
"GPU": 1
},
"sample_packing": false,
"sample_packing_bin_size": 200,
"sample_packing_group_size": 100000,
"save_only_model": false,
"save_safetensors": true,
"save_steps": 100,
"save_strategy": "steps",
"sequence_len": 8192,
"shuffle_before_merging_datasets": false,
"shuffle_merged_datasets": true,
"skip_prepare_dataset": false,
"streaming_multipack_buffer_size": 10000,
"strict": false,
"tensor_parallel_size": 1,
"tf32": true,
"tiled_mlp_use_original_mlp": true,
"tokenizer_config": "Qwen/Qwen2.5-Coder-14B-Instruct",
"tokenizer_save_jinja_files": true,
"torch_dtype": "torch.bfloat16",
"train_on_inputs": false,
"trl": {
"log_completions": false,
"mask_truncated_completions": false,
"ref_model_mixup_alpha": 0.9,
"ref_model_sync_steps": 64,
"scale_rewards": true,
"sync_ref_model": false,
"use_vllm": false,
"vllm_server_host": "0.0.0.0",
"vllm_server_port": 8000
},
"use_ray": false,
"val_set_size": 0.0,
"vllm": {
"device": "auto",
"dtype": "auto",
"gpu_memory_utilization": 0.9,
"host": "0.0.0.0",
"port": 8000
},
"warmup_ratio": 0.1,
"weight_decay": 0.01,
"world_size": 1
}
[2025-12-25 22:28:00,313] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:278] [PID:1133] EOS: 151645 / <|im_end|>
[2025-12-25 22:28:00,314] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:279] [PID:1133] BOS: None / None
[2025-12-25 22:28:00,314] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:280] [PID:1133] PAD: 151643 / <|endoftext|>
[2025-12-25 22:28:00,314] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:281] [PID:1133] UNK: None / None
[2025-12-25 22:28:00,314] [INFO] [axolotl.utils.data.shared.load_preprocessed_dataset:476] [PID:1133] Unable to find prepared dataset in last_run_prepared/fd30d23b351de719c91e124efcc5fe43
[2025-12-25 22:28:00,314] [INFO] [axolotl.utils.data.sft._load_raw_datasets:320] [PID:1133] Loading raw datasets...
[2025-12-25 22:28:00,314] [WARNING] [axolotl.utils.data.sft._load_raw_datasets:322] [PID:1133] Processing datasets during training can lead to VRAM instability. Please pre-process your dataset using `axolotl preprocess path/to/config.yml`.
[2025-12-25 22:28:00,955] [INFO] [axolotl.utils.data.wrappers.get_dataset_wrapper:87] [PID:1133] Loading dataset: mbakgun/n8nbuilder-n8n-workflows-dataset with base_type: alpaca and prompt_style: None
[2025-12-25 22:28:01,168] [INFO] [axolotl.utils.data.utils.handle_long_seq_in_dataset:218] [PID:1133] min_input_len: 878
[2025-12-25 22:28:01,168] [INFO] [axolotl.utils.data.utils.handle_long_seq_in_dataset:220] [PID:1133] max_input_len: 12396
Dropping Long Sequences (>8192) (num_proc=36): 0%| | 0/2737 [00:00<?, ? examples/s]
Dropping Long Sequences (>8192) (num_proc=36): 3%|██▎ | 77/2737 [00:00<00:31, 85.60 examples/s]
Dropping Long Sequences (>8192) (num_proc=36): 42%|████████████████████████████████▉ | 1141/2737 [00:00<00:01, 1531.24 examples/s]
Dropping Long Sequences (>8192) (num_proc=36): 100%|███████████████████████████████████████████████████████████████████████████████| 2737/2737 [00:01<00:00, 3888.53 examples/s]
Dropping Long Sequences (>8192) (num_proc=36): 100%|███████████████████████████████████████████████████████████████████████████████| 2737/2737 [00:01<00:00, 2115.72 examples/s]
[2025-12-25 22:28:02,540] [WARNING] [axolotl.utils.data.utils.handle_long_seq_in_dataset:260] [PID:1133] Dropped 433 samples from dataset
Saving the dataset (0/9 shards): 0%| | 0/2304 [00:00<?, ? examples/s]
Saving the dataset (0/9 shards): 11%|██████████▌ | 256/2304 [00:00<00:02, 916.49 examples/s]
Saving the dataset (1/9 shards): 11%|██████████▌ | 256/2304 [00:00<00:02, 916.49 examples/s]
Saving the dataset (2/9 shards): 33%|███████████████████████████████▋ | 768/2304 [00:00<00:01, 916.49 examples/s]
Saving the dataset (3/9 shards): 33%|███████████████████████████████▋ | 768/2304 [00:00<00:01, 916.49 examples/s]
Saving the dataset (4/9 shards): 44%|█████████████████████████████████████████▊ | 1024/2304 [00:00<00:01, 916.49 examples/s]
Saving the dataset (5/9 shards): 56%|████████████████████████████████████████████████████▏ | 1280/2304 [00:00<00:01, 916.49 examples/s]
Saving the dataset (6/9 shards): 67%|██████████████████████████████████████████████████████████████▋ | 1536/2304 [00:00<00:00, 916.49 examples/s]
Saving the dataset (7/9 shards): 78%|█████████████████████████████████████████████████████████████████████████ | 1792/2304 [00:00<00:00, 916.49 examples/s]
Saving the dataset (8/9 shards): 89%|███████████████████████████████████████████████████████████████████████████████████▌ | 2048/2304 [00:00<00:00, 916.49 examples/s]
Saving the dataset (9/9 shards): 100%|██████████████████████████████████████████████████████████████████████████████████████████████| 2304/2304 [00:00<00:00, 916.49 examples/s]
Saving the dataset (9/9 shards): 100%|█████████████████████████████████████████████████████████████████████████████████████████████| 2304/2304 [00:00<00:00, 5976.92 examples/s]
[2025-12-25 22:28:03,194] [DEBUG] [axolotl.utils.trainer.calculate_total_num_steps:404] [PID:1133] total_num_tokens: 9_507_792
[2025-12-25 22:28:03,238] [DEBUG] [axolotl.utils.trainer.calculate_total_num_steps:422] [PID:1133] `total_supervised_tokens: 11_572_652`
[2025-12-25 22:28:03,239] [DEBUG] [axolotl.utils.trainer.calculate_total_num_steps:520] [PID:1133] total_num_steps: 432
[2025-12-25 22:28:03,239] [INFO] [axolotl.utils.data.sft._prepare_standard_dataset:121] [PID:1133] Maximum number of steps set at 432
[2025-12-25 22:28:03,267] [DEBUG] [axolotl.train.setup_model_and_tokenizer:65] [PID:1133] Loading tokenizer... Qwen/Qwen2.5-Coder-14B-Instruct
[2025-12-25 22:28:03,684] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:278] [PID:1133] EOS: 151645 / <|im_end|>
[2025-12-25 22:28:03,684] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:279] [PID:1133] BOS: None / None
[2025-12-25 22:28:03,685] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:280] [PID:1133] PAD: 151643 / <|endoftext|>
[2025-12-25 22:28:03,685] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:281] [PID:1133] UNK: None / None
[2025-12-25 22:28:03,685] [DEBUG] [axolotl.train.setup_model_and_tokenizer:74] [PID:1133] Loading model
[2025-12-25 22:28:03,736] [DEBUG] [axolotl.monkeypatch.transformers.trainer_loss_calc.patch_evaluation_loop:87] [PID:1133] Patched Trainer.evaluation_loop with nanmean loss calculation
[2025-12-25 22:28:03,737] [DEBUG] [axolotl.monkeypatch.transformers.trainer_loss_calc.patch_maybe_log_save_evaluate:138] [PID:1133] Patched Trainer._maybe_log_save_evaluate with nanmean loss calculation
Loading checkpoint shards: 0%| | 0/6 [00:00<?, ?it/s]
Loading checkpoint shards: 17%|███████████████████ | 1/6 [00:03<00:19, 3.93s/it]
Loading checkpoint shards: 33%|██████████████████████████████████████ | 2/6 [00:09<00:18, 4.62s/it]
Loading checkpoint shards: 50%|█████████████████████████████████████████████████████████ | 3/6 [00:14<00:14, 4.80s/it]
Loading checkpoint shards: 67%|████████████████████████████████████████████████████████████████████████████ | 4/6 [00:19<00:09, 4.89s/it]
Loading checkpoint shards: 83%|███████████████████████████████████████████████████████████████████████████████████████████████ | 5/6 [00:24<00:04, 4.91s/it]
Loading checkpoint shards: 100%|██████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 6/6 [00:27<00:00, 4.42s/it]
Loading checkpoint shards: 100%|██████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 6/6 [00:27<00:00, 4.58s/it]
[2025-12-25 22:28:32,092] [INFO] [axolotl.loaders.model._prepare_model_for_quantization:863] [PID:1133] converting PEFT model w/ prepare_model_for_kbit_training
[2025-12-25 22:28:32,098] [INFO] [axolotl.loaders.model._configure_embedding_dtypes:345] [PID:1133] Converting modules to torch.bfloat16
[2025-12-25 22:28:32,100] [DEBUG] [axolotl.loaders.model.log_gpu_memory_usage:127] [PID:1133] Memory usage after model load 13.888GB (+13.888GB allocated, +15.756GB reserved)
trainable params: 137,625,600 || all params: 14,907,659,264 || trainable%: 0.9232
[2025-12-25 22:28:33,795] [DEBUG] [axolotl.loaders.model.log_gpu_memory_usage:127] [PID:1133] after adapters 9.900GB (+9.900GB allocated, +16.031GB reserved)
[2025-12-25 22:28:40,857] [INFO] [axolotl.train.save_initial_configs:398] [PID:1133] Pre-saving adapter config to ./outputs/qwen25-coder-n8n...
[2025-12-25 22:28:40,857] [INFO] [axolotl.train.save_initial_configs:402] [PID:1133] Pre-saving tokenizer to ./outputs/qwen25-coder-n8n...
[2025-12-25 22:28:41,011] [INFO] [axolotl.train.save_initial_configs:407] [PID:1133] Pre-saving model config to ./outputs/qwen25-coder-n8n...
[2025-12-25 22:28:41,013] [INFO] [axolotl.train.execute_training:196] [PID:1133] Starting trainer...
0%| | 0/432 [00:00<?, ?it/s]
0%|▎ | 1/432 [00:33<4:03:10, 33.85s/it]
{'loss': 1.0821, 'grad_norm': 0.09848134219646454, 'learning_rate': 0.0, 'memory/max_active (GiB)': 27.69, 'memory/max_allocated (GiB)': 27.69, 'memory/device_reserved (GiB)': 30.8, 'tokens_per_second_per_gpu': 1763.68, 'epoch': 0.01}
0%|▎ | 1/432 [00:33<4:03:10, 33.85s/it]
0%|▋ | 2/432 [01:03<3:44:28, 31.32s/it]
{'loss': 1.2119, 'grad_norm': 0.11234692484140396, 'learning_rate': 4.651162790697674e-06, 'memory/max_active (GiB)': 25.68, 'memory/max_allocated (GiB)': 25.68, 'memory/device_reserved (GiB)': 31.05, 'tokens_per_second_per_gpu': 1834.64, 'epoch': 0.01}
0%|▋ | 2/432 [01:03<3:44:28, 31.32s/it]
1%|▉ | 3/432 [01:33<3:40:04, 30.78s/it]
{'loss': 1.2053, 'grad_norm': 0.11071926355361938, 'learning_rate': 9.302325581395349e-06, 'memory/max_active (GiB)': 27.01, 'memory/max_allocated (GiB)': 27.01, 'memory/device_reserved (GiB)': 31.05, 'tokens_per_second_per_gpu': 1875.82, 'epoch': 0.02}
1%|▉ | 3/432 [01:33<3:40:04, 30.78s/it]
1%|█▎ | 4/432 [02:07<3:49:36, 32.19s/it]
{'loss': 1.0514, 'grad_norm': 0.10147764533758163, 'learning_rate': 1.3953488372093024e-05, 'memory/max_active (GiB)': 28.9, 'memory/max_allocated (GiB)': 28.9, 'memory/device_reserved (GiB)': 32.27, 'tokens_per_second_per_gpu': 1792.72, 'epoch': 0.03}
1%|█▎ | 4/432 [02:07<3:49:36, 32.19s/it]
1%|█▌ | 5/432 [02:41<3:51:32, 32.54s/it]
{'loss': 1.209, 'grad_norm': 0.10568977892398834, 'learning_rate': 1.8604651162790697e-05, 'memory/max_active (GiB)': 27.95, 'memory/max_allocated (GiB)': 27.95, 'memory/device_reserved (GiB)': 32.27, 'tokens_per_second_per_gpu': 1840.61, 'epoch': 0.03}
1%|█▌ | 5/432 [02:41<3:51:32, 32.54s/it]
1%|█▉ | 6/432 [03:13<3:51:42, 32.63s/it]
{'loss': 1.0817, 'grad_norm': 0.10363873094320297, 'learning_rate': 2.3255813953488374e-05, 'memory/max_active (GiB)': 27.95, 'memory/max_allocated (GiB)': 27.95, 'memory/device_reserved (GiB)': 32.27, 'tokens_per_second_per_gpu': 1784.92, 'epoch': 0.04}
1%|█▉ | 6/432 [03:13<3:51:42, 32.63s/it]
2%|██▏ | 7/432 [03:52<4:04:08, 34.47s/it]
{'loss': 1.1571, 'grad_norm': 0.113986074924469, 'learning_rate': 2.7906976744186048e-05, 'memory/max_active (GiB)': 28.9, 'memory/max_allocated (GiB)': 28.9, 'memory/device_reserved (GiB)': 32.33, 'tokens_per_second_per_gpu': 1892.63, 'epoch': 0.05}
2%|██▏ | 7/432 [03:52<4:04:08, 34.47s/it]
2%|██▌ | 8/432 [04:26<4:02:41, 34.34s/it]
{'loss': 1.1444, 'grad_norm': 0.1191892921924591, 'learning_rate': 3.2558139534883724e-05, 'memory/max_active (GiB)': 28.9, 'memory/max_allocated (GiB)': 28.9, 'memory/device_reserved (GiB)': 32.33, 'tokens_per_second_per_gpu': 1863.93, 'epoch': 0.06}
2%|██▌ | 8/432 [04:26<4:02:41, 34.34s/it]
2%|██▊ | 9/432 [04:54<3:49:22, 32.53s/it]
{'loss': 1.1786, 'grad_norm': 0.11628979444503784, 'learning_rate': 3.7209302325581394e-05, 'memory/max_active (GiB)': 24.64, 'memory/max_allocated (GiB)': 24.64, 'memory/device_reserved (GiB)': 32.33, 'tokens_per_second_per_gpu': 1764.21, 'epoch': 0.06}
2%|██▊ | 9/432 [04:54<3:49:22, 32.53s/it]
2%|███▏ | 10/432 [05:25<3:44:45, 31.96s/it]
{'loss': 1.0695, 'grad_norm': 0.10155434161424637, 'learning_rate': 4.186046511627907e-05, 'memory/max_active (GiB)': 27.2, 'memory/max_allocated (GiB)': 27.2, 'memory/device_reserved (GiB)': 32.33, 'tokens_per_second_per_gpu': 1823.2, 'epoch': 0.07}
2%|███▏ | 10/432 [05:25<3:44:45, 31.96s/it]
3%|███▍ | 11/432 [06:01<3:52:59, 33.21s/it]
{'loss': 1.0805, 'grad_norm': 0.08485760539770126, 'learning_rate': 4.651162790697675e-05, 'memory/max_active (GiB)': 27.01, 'memory/max_allocated (GiB)': 27.01, 'memory/device_reserved (GiB)': 32.33, 'tokens_per_second_per_gpu': 1805.0, 'epoch': 0.08}
3%|███▍ | 11/432 [06:01<3:52:59, 33.21s/it]
3%|███▊ | 12/432 [06:30<3:44:23, 32.05s/it]
{'loss': 1.0824, 'grad_norm': 0.07211048156023026, 'learning_rate': 5.1162790697674425e-05, 'memory/max_active (GiB)': 27.95, 'memory/max_allocated (GiB)': 27.95, 'memory/device_reserved (GiB)': 32.33, 'tokens_per_second_per_gpu': 1799.74, 'epoch': 0.08}
3%|███▊ | 12/432 [06:30<3:44:23, 32.05s/it]
3%|████ | 13/432 [07:04<3:47:30, 32.58s/it]
{'loss': 1.0264, 'grad_norm': 0.06483420729637146, 'learning_rate': 5.5813953488372095e-05, 'memory/max_active (GiB)': 27.95, 'memory/max_allocated (GiB)': 27.95, 'memory/device_reserved (GiB)': 32.33, 'tokens_per_second_per_gpu': 1798.61, 'epoch': 0.09}
3%|████ | 13/432 [07:04<3:47:30, 32.58s/it]
3%|████▍ | 14/432 [07:32<3:37:27, 31.21s/it]
{'loss': 1.0967, 'grad_norm': 0.06657296419143677, 'learning_rate': 6.0465116279069765e-05, 'memory/max_active (GiB)': 28.9, 'memory/max_allocated (GiB)': 28.9, 'memory/device_reserved (GiB)': 32.33, 'tokens_per_second_per_gpu': 1788.44, 'epoch': 0.1}
3%|████▍ | 14/432 [07:32<3:37:27, 31.21s/it]
3%|████▋ | 15/432 [07:59<3:26:38, 29.73s/it]
{'loss': 1.1489, 'grad_norm': 0.195042684674263, 'learning_rate': 6.511627906976745e-05, 'memory/max_active (GiB)': 24.8, 'memory/max_allocated (GiB)': 24.8, 'memory/device_reserved (GiB)': 32.33, 'tokens_per_second_per_gpu': 1785.21, 'epoch': 0.1}
3%|████▋ | 15/432 [07:59<3:26:38, 29.73s/it]
4%|█████ | 16/432 [08:26<3:21:43, 29.10s/it]
{'loss': 1.0985, 'grad_norm': 0.07728952169418335, 'learning_rate': 6.976744186046513e-05, 'memory/max_active (GiB)': 27.95, 'memory/max_allocated (GiB)': 27.95, 'memory/device_reserved (GiB)': 32.33, 'tokens_per_second_per_gpu': 1752.91, 'epoch': 0.11}
4%|█████ | 16/432 [08:26<3:21:43, 29.10s/it]
4%|█████▎ | 17/432 [08:55<3:20:08, 28.93s/it]
{'loss': 1.1112, 'grad_norm': 0.08134876191616058, 'learning_rate': 7.441860465116279e-05, 'memory/max_active (GiB)': 26.06, 'memory/max_allocated (GiB)': 26.06, 'memory/device_reserved (GiB)': 32.33, 'tokens_per_second_per_gpu': 1813.48, 'epoch': 0.12}
4%|█████▎ | 17/432 [08:55<3:20:08, 28.93s/it]
4%|█████▋ | 18/432 [09:24<3:20:20, 29.04s/it]
{'loss': 1.0222, 'grad_norm': 0.08289807289838791, 'learning_rate': 7.906976744186047e-05, 'memory/max_active (GiB)': 27.95, 'memory/max_allocated (GiB)': 27.95, 'memory/device_reserved (GiB)': 32.33, 'tokens_per_second_per_gpu': 1852.47, 'epoch': 0.12}
4%|█████▋ | 18/432 [09:24<3:20:20, 29.04s/it]
4%|█████▉ | 19/432 [09:50<3:12:40, 27.99s/it]
{'loss': 1.1493, 'grad_norm': 0.09635733813047409, 'learning_rate': 8.372093023255814e-05, 'memory/max_active (GiB)': 24.17, 'memory/max_allocated (GiB)': 24.17, 'memory/device_reserved (GiB)': 32.33, 'tokens_per_second_per_gpu': 1745.06, 'epoch': 0.13}
4%|█████▉ | 19/432 [09:50<3:12:40, 27.99s/it]
5%|██████▎ | 20/432 [10:16<3:09:20, 27.57s/it]
{'loss': 0.9912, 'grad_norm': 0.08602173626422882, 'learning_rate': 8.837209302325582e-05, 'memory/max_active (GiB)': 26.25, 'memory/max_allocated (GiB)': 26.25, 'memory/device_reserved (GiB)': 32.33, 'tokens_per_second_per_gpu': 1758.58, 'epoch': 0.14}
5%|██████▎ | 20/432 [10:16<3:09:20, 27.57s/it]
5%|██████▌ | 21/432 [10:45<3:11:39, 27.98s/it]
{'loss': 1.1637, 'grad_norm': 0.08320974558591843, 'learning_rate': 9.30232558139535e-05, 'memory/max_active (GiB)': 27.95, 'memory/max_allocated (GiB)': 27.95, 'memory/device_reserved (GiB)': 32.33, 'tokens_per_second_per_gpu': 1772.58, 'epoch': 0.15}
5%|██████▌ | 21/432 [10:45<3:11:39, 27.98s/it]
5%|██████▉ | 22/432 [11:22<3:29:56, 30.72s/it]
{'loss': 1.0209, 'grad_norm': 0.0785663053393364, 'learning_rate': 9.767441860465116e-05, 'memory/max_active (GiB)': 28.9, 'memory/max_allocated (GiB)': 28.9, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1822.66, 'epoch': 0.15}
5%|██████▉ | 22/432 [11:22<3:29:56, 30.72s/it]
5%|███████▏ | 23/432 [11:55<3:33:42, 31.35s/it]
{'loss': 0.9858, 'grad_norm': 0.07734047621488571, 'learning_rate': 0.00010232558139534885, 'memory/max_active (GiB)': 27.95, 'memory/max_allocated (GiB)': 27.95, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1889.88, 'epoch': 0.16}
5%|███████▏ | 23/432 [11:55<3:33:42, 31.35s/it]
6%|███████▌ | 24/432 [12:27<3:35:25, 31.68s/it]
{'loss': 1.0003, 'grad_norm': 0.07255646586418152, 'learning_rate': 0.00010697674418604651, 'memory/max_active (GiB)': 28.9, 'memory/max_allocated (GiB)': 28.9, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1833.16, 'epoch': 0.17}
6%|███████▌ | 24/432 [12:27<3:35:25, 31.68s/it]
6%|███████▊ | 25/432 [13:04<3:45:06, 33.19s/it]
{'loss': 1.0143, 'grad_norm': 0.07897679507732391, 'learning_rate': 0.00011162790697674419, 'memory/max_active (GiB)': 28.15, 'memory/max_allocated (GiB)': 28.15, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1900.98, 'epoch': 0.17}
6%|███████▊ | 25/432 [13:04<3:45:06, 33.19s/it]
6%|████████▏ | 26/432 [13:29<3:28:01, 30.74s/it]
{'loss': 1.0341, 'grad_norm': 0.09510312229394913, 'learning_rate': 0.00011627906976744187, 'memory/max_active (GiB)': 23.18, 'memory/max_allocated (GiB)': 23.18, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1701.34, 'epoch': 0.18}
6%|████████▏ | 26/432 [13:29<3:28:01, 30.74s/it]
6%|████████▌ | 27/432 [14:03<3:33:11, 31.58s/it]
{'loss': 1.0004, 'grad_norm': 0.07016909122467041, 'learning_rate': 0.00012093023255813953, 'memory/max_active (GiB)': 28.38, 'memory/max_allocated (GiB)': 28.38, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1891.5, 'epoch': 0.19}
6%|████████▌ | 27/432 [14:03<3:33:11, 31.58s/it]
6%|████████▊ | 28/432 [14:40<3:45:02, 33.42s/it]
{'loss': 0.9588, 'grad_norm': 0.07151541113853455, 'learning_rate': 0.0001255813953488372, 'memory/max_active (GiB)': 28.9, 'memory/max_allocated (GiB)': 28.9, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1890.08, 'epoch': 0.19}
6%|████████▊ | 28/432 [14:40<3:45:02, 33.42s/it]
7%|█████████▏ | 29/432 [15:12<3:39:52, 32.74s/it]
{'loss': 1.0078, 'grad_norm': 0.07155290246009827, 'learning_rate': 0.0001302325581395349, 'memory/max_active (GiB)': 27.01, 'memory/max_allocated (GiB)': 27.01, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1759.11, 'epoch': 0.2}
7%|█████████▏ | 29/432 [15:12<3:39:52, 32.74s/it]
7%|█████████▍ | 30/432 [15:39<3:27:49, 31.02s/it]
{'loss': 1.0343, 'grad_norm': 0.08267220109701157, 'learning_rate': 0.00013488372093023256, 'memory/max_active (GiB)': 26.49, 'memory/max_allocated (GiB)': 26.49, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1794.99, 'epoch': 0.21}
7%|█████████▍ | 30/432 [15:39<3:27:49, 31.02s/it]
7%|█████████▊ | 31/432 [16:13<3:34:58, 32.17s/it]
{'loss': 0.9155, 'grad_norm': 0.06379543989896774, 'learning_rate': 0.00013953488372093025, 'memory/max_active (GiB)': 28.9, 'memory/max_allocated (GiB)': 28.9, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1829.94, 'epoch': 0.22}
7%|█████████▊ | 31/432 [16:13<3:34:58, 32.17s/it]
7%|██████████ | 32/432 [16:40<3:23:21, 30.50s/it]
{'loss': 1.0727, 'grad_norm': 0.07846751064062119, 'learning_rate': 0.00014418604651162791, 'memory/max_active (GiB)': 23.79, 'memory/max_allocated (GiB)': 23.79, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1854.31, 'epoch': 0.22}
7%|██████████ | 32/432 [16:40<3:23:21, 30.50s/it]
8%|██████████▍ | 33/432 [17:11<3:23:04, 30.54s/it]
{'loss': 1.0472, 'grad_norm': 0.07601239532232285, 'learning_rate': 0.00014883720930232558, 'memory/max_active (GiB)': 28.9, 'memory/max_allocated (GiB)': 28.9, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1806.02, 'epoch': 0.23}
8%|██████████▍ | 33/432 [17:11<3:23:04, 30.54s/it]
8%|██████████▋ | 34/432 [17:38<3:15:19, 29.45s/it]
{'loss': 1.034, 'grad_norm': 0.09074926376342773, 'learning_rate': 0.00015348837209302327, 'memory/max_active (GiB)': 23.79, 'memory/max_allocated (GiB)': 23.79, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1814.81, 'epoch': 0.24}
8%|██████████▋ | 34/432 [17:38<3:15:19, 29.45s/it]
8%|███████████ | 35/432 [18:09<3:19:04, 30.09s/it]
{'loss': 0.9786, 'grad_norm': 0.07441543787717819, 'learning_rate': 0.00015813953488372093, 'memory/max_active (GiB)': 27.01, 'memory/max_allocated (GiB)': 27.01, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1865.18, 'epoch': 0.24}
8%|███████████ | 35/432 [18:09<3:19:04, 30.09s/it]
8%|███████████▎ | 36/432 [18:43<3:26:48, 31.33s/it]
{'loss': 0.9218, 'grad_norm': 0.08436308056116104, 'learning_rate': 0.00016279069767441862, 'memory/max_active (GiB)': 27.01, 'memory/max_allocated (GiB)': 27.01, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1860.2, 'epoch': 0.25}
8%|███████████▎ | 36/432 [18:43<3:26:48, 31.33s/it]
9%|███████████▋ | 37/432 [19:13<3:22:58, 30.83s/it]
{'loss': 1.0504, 'grad_norm': 0.07554468512535095, 'learning_rate': 0.00016744186046511629, 'memory/max_active (GiB)': 25.68, 'memory/max_allocated (GiB)': 25.68, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1816.31, 'epoch': 0.26}
9%|███████████▋ | 37/432 [19:13<3:22:58, 30.83s/it]
9%|███████████▉ | 38/432 [19:44<3:22:43, 30.87s/it]
{'loss': 0.9141, 'grad_norm': 0.09911656379699707, 'learning_rate': 0.00017209302325581395, 'memory/max_active (GiB)': 26.26, 'memory/max_allocated (GiB)': 26.26, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1863.8, 'epoch': 0.26}
9%|███████████▉ | 38/432 [19:44<3:22:43, 30.87s/it]
9%|████████████▎ | 39/432 [20:06<3:03:55, 28.08s/it]
{'loss': 1.0342, 'grad_norm': 0.07778877764940262, 'learning_rate': 0.00017674418604651164, 'memory/max_active (GiB)': 24.17, 'memory/max_allocated (GiB)': 24.17, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1619.33, 'epoch': 0.27}
9%|████████████▎ | 39/432 [20:06<3:03:55, 28.08s/it]
9%|████████████▌ | 40/432 [20:35<3:05:18, 28.36s/it]
{'loss': 0.9365, 'grad_norm': 0.09776122868061066, 'learning_rate': 0.0001813953488372093, 'memory/max_active (GiB)': 26.06, 'memory/max_allocated (GiB)': 26.06, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1708.88, 'epoch': 0.28}
9%|████████████▌ | 40/432 [20:35<3:05:18, 28.36s/it]
9%|████████████▉ | 41/432 [21:07<3:12:06, 29.48s/it]
{'loss': 0.9455, 'grad_norm': 0.09212527424097061, 'learning_rate': 0.000186046511627907, 'memory/max_active (GiB)': 28.9, 'memory/max_allocated (GiB)': 28.9, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1854.36, 'epoch': 0.28}
9%|████████████▉ | 41/432 [21:07<3:12:06, 29.48s/it]
10%|█████████████▏ | 42/432 [21:32<3:03:44, 28.27s/it]
{'loss': 0.9964, 'grad_norm': 0.1160384938120842, 'learning_rate': 0.00019069767441860466, 'memory/max_active (GiB)': 23.7, 'memory/max_allocated (GiB)': 23.7, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1775.41, 'epoch': 0.29}
10%|█████████████▏ | 42/432 [21:32<3:03:44, 28.27s/it]
10%|█████████████▌ | 43/432 [22:02<3:06:18, 28.74s/it]
{'loss': 0.8627, 'grad_norm': 0.06805545091629028, 'learning_rate': 0.00019534883720930232, 'memory/max_active (GiB)': 25.68, 'memory/max_allocated (GiB)': 25.68, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1830.25, 'epoch': 0.3}
10%|█████████████▌ | 43/432 [22:02<3:06:18, 28.74s/it]
10%|█████████████▊ | 44/432 [22:35<3:13:22, 29.90s/it]
{'loss': 0.9417, 'grad_norm': 0.06951376795768738, 'learning_rate': 0.0002, 'memory/max_active (GiB)': 25.68, 'memory/max_allocated (GiB)': 25.68, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1843.05, 'epoch': 0.31}
10%|█████████████▊ | 44/432 [22:35<3:13:22, 29.90s/it]
10%|██████████████▏ | 45/432 [23:07<3:18:05, 30.71s/it]
{'loss': 0.9017, 'grad_norm': 0.06728649139404297, 'learning_rate': 0.00019999673886943734, 'memory/max_active (GiB)': 27.95, 'memory/max_allocated (GiB)': 27.95, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1889.73, 'epoch': 0.31}
10%|██████████████▏ | 45/432 [23:07<3:18:05, 30.71s/it]
11%|██████████████▍ | 46/432 [23:39<3:19:09, 30.96s/it]
{'loss': 1.0087, 'grad_norm': 0.0888209342956543, 'learning_rate': 0.0001999869556904488, 'memory/max_active (GiB)': 27.95, 'memory/max_allocated (GiB)': 27.95, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1805.58, 'epoch': 0.32}
11%|██████████████▍ | 46/432 [23:39<3:19:09, 30.96s/it]
11%|██████████████▊ | 47/432 [24:08<3:16:18, 30.59s/it]
{'loss': 0.9246, 'grad_norm': 0.07477093487977982, 'learning_rate': 0.00019997065110111885, 'memory/max_active (GiB)': 27.2, 'memory/max_allocated (GiB)': 27.2, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1807.73, 'epoch': 0.33}
11%|██████████████▊ | 47/432 [24:08<3:16:18, 30.59s/it]
11%|███████████████ | 48/432 [24:39<3:15:53, 30.61s/it]
{'loss': 0.936, 'grad_norm': 0.08000776916742325, 'learning_rate': 0.00019994782616487538, 'memory/max_active (GiB)': 27.01, 'memory/max_allocated (GiB)': 27.01, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1732.16, 'epoch': 0.33}
11%|███████████████ | 48/432 [24:39<3:15:53, 30.61s/it]
11%|███████████████▍ | 49/432 [25:06<3:08:03, 29.46s/it]
{'loss': 0.9732, 'grad_norm': 0.2703610956668854, 'learning_rate': 0.00019991848237042035, 'memory/max_active (GiB)': 24.64, 'memory/max_allocated (GiB)': 24.64, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1814.28, 'epoch': 0.34}
11%|███████████████▍ | 49/432 [25:06<3:08:03, 29.46s/it]
12%|███████████████▋ | 50/432 [25:34<3:04:18, 28.95s/it]
{'loss': 0.9867, 'grad_norm': 0.08173573762178421, 'learning_rate': 0.00019988262163163264, 'memory/max_active (GiB)': 26.25, 'memory/max_allocated (GiB)': 26.25, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1746.16, 'epoch': 0.35}
12%|███████████████▋ | 50/432 [25:34<3:04:18, 28.95s/it]
12%|████████████████ | 51/432 [26:06<3:10:51, 30.06s/it]
{'loss': 0.9353, 'grad_norm': 0.06703449040651321, 'learning_rate': 0.00019984024628744328, 'memory/max_active (GiB)': 25.31, 'memory/max_allocated (GiB)': 25.31, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1803.49, 'epoch': 0.35}
12%|████████████████ | 51/432 [26:06<3:10:51, 30.06s/it]
12%|████████████████▎ | 52/432 [26:31<3:00:46, 28.54s/it]
{'loss': 0.9705, 'grad_norm': 0.0770621970295906, 'learning_rate': 0.0001997913591016829, 'memory/max_active (GiB)': 27.44, 'memory/max_allocated (GiB)': 27.44, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1805.23, 'epoch': 0.36}
12%|████████████████▎ | 52/432 [26:31<3:00:46, 28.54s/it]
12%|████████████████▋ | 53/432 [27:04<3:08:21, 29.82s/it]
{'loss': 0.9082, 'grad_norm': 0.08800782263278961, 'learning_rate': 0.00019973596326290137, 'memory/max_active (GiB)': 28.9, 'memory/max_allocated (GiB)': 28.9, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1787.79, 'epoch': 0.37}
12%|████████████████▋ | 53/432 [27:04<3:08:21, 29.82s/it]
12%|█████████████████ | 54/432 [27:38<3:16:01, 31.11s/it]
{'loss': 0.964, 'grad_norm': 0.0656328946352005, 'learning_rate': 0.00019967406238415998, 'memory/max_active (GiB)': 28.9, 'memory/max_allocated (GiB)': 28.9, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1888.13, 'epoch': 0.38}
12%|█████████████████ | 54/432 [27:38<3:16:01, 31.11s/it]
13%|█████████████████▎ | 55/432 [28:08<3:13:44, 30.84s/it]
{'loss': 0.918, 'grad_norm': 0.09178014099597931, 'learning_rate': 0.00019960566050279566, 'memory/max_active (GiB)': 27.2, 'memory/max_allocated (GiB)': 27.2, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1822.06, 'epoch': 0.38}
13%|█████████████████▎ | 55/432 [28:08<3:13:44, 30.84s/it]
13%|█████████████████▋ | 56/432 [28:34<3:03:56, 29.35s/it]
{'loss': 1.0078, 'grad_norm': 0.07544898241758347, 'learning_rate': 0.00019953076208015772, 'memory/max_active (GiB)': 28.9, 'memory/max_allocated (GiB)': 28.9, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1779.76, 'epoch': 0.39}
13%|█████████████████▋ | 56/432 [28:34<3:03:56, 29.35s/it]
13%|█████████████████▉ | 57/432 [29:07<3:09:09, 30.27s/it]
{'loss': 0.952, 'grad_norm': 0.07013165950775146, 'learning_rate': 0.0001994493720013169, 'memory/max_active (GiB)': 27.01, 'memory/max_allocated (GiB)': 27.01, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1875.9, 'epoch': 0.4}
13%|█████████████████▉ | 57/432 [29:07<3:09:09, 30.27s/it]
13%|██████████████████▎ | 58/432 [29:36<3:06:17, 29.89s/it]
{'loss': 0.9751, 'grad_norm': 0.2212851643562317, 'learning_rate': 0.00019936149557474666, 'memory/max_active (GiB)': 27.01, 'memory/max_allocated (GiB)': 27.01, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1808.54, 'epoch': 0.4}
13%|██████████████████▎ | 58/432 [29:36<3:06:17, 29.89s/it]
14%|██████████████████▌ | 59/432 [30:09<3:12:19, 30.94s/it]
{'loss': 0.9055, 'grad_norm': 0.0661102756857872, 'learning_rate': 0.00019926713853197695, 'memory/max_active (GiB)': 24.74, 'memory/max_allocated (GiB)': 24.74, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1802.3, 'epoch': 0.41}
14%|██████████████████▌ | 59/432 [30:09<3:12:19, 30.94s/it]
14%|██████████████████▉ | 60/432 [30:40<3:12:31, 31.05s/it]
{'loss': 0.9826, 'grad_norm': 0.08663811534643173, 'learning_rate': 0.0001991663070272206, 'memory/max_active (GiB)': 24.17, 'memory/max_allocated (GiB)': 24.17, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1707.88, 'epoch': 0.42}
14%|██████████████████▉ | 60/432 [30:40<3:12:31, 31.05s/it]
14%|███████████████████▏ | 61/432 [31:14<3:15:52, 31.68s/it]
{'loss': 0.9759, 'grad_norm': 0.07683200389146805, 'learning_rate': 0.0001990590076369715, 'memory/max_active (GiB)': 27.95, 'memory/max_allocated (GiB)': 27.95, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1895.98, 'epoch': 0.42}
14%|███████████████████▏ | 61/432 [31:14<3:15:52, 31.68s/it]
14%|███████████████████▌ | 62/432 [31:45<3:14:33, 31.55s/it]
{'loss': 0.9168, 'grad_norm': 0.07595925778150558, 'learning_rate': 0.00019894524735957622, 'memory/max_active (GiB)': 28.9, 'memory/max_allocated (GiB)': 28.9, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1797.18, 'epoch': 0.43}
14%|███████████████████▌ | 62/432 [31:45<3:14:33, 31.55s/it]
15%|███████████████████▊ | 63/432 [32:13<3:08:40, 30.68s/it]
{'loss': 0.9679, 'grad_norm': 0.07661418616771698, 'learning_rate': 0.00019882503361477705, 'memory/max_active (GiB)': 28.15, 'memory/max_allocated (GiB)': 28.15, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1830.69, 'epoch': 0.44}
15%|███████████████████▊ | 63/432 [32:13<3:08:40, 30.68s/it]
15%|████████████████████▏ | 64/432 [32:44<3:08:27, 30.73s/it]
{'loss': 0.9592, 'grad_norm': 0.08054457604885101, 'learning_rate': 0.00019869837424322829, 'memory/max_active (GiB)': 27.95, 'memory/max_allocated (GiB)': 27.95, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1798.08, 'epoch': 0.44}
15%|████████████████████▏ | 64/432 [32:44<3:08:27, 30.73s/it]
15%|████████████████████▍ | 65/432 [33:15<3:08:07, 30.76s/it]
{'loss': 0.9257, 'grad_norm': 0.08320043236017227, 'learning_rate': 0.00019856527750598493, 'memory/max_active (GiB)': 28.15, 'memory/max_allocated (GiB)': 28.15, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1795.23, 'epoch': 0.45}
15%|████████████████████▍ | 65/432 [33:15<3:08:07, 30.76s/it]
15%|████████████████████▊ | 66/432 [33:49<3:13:58, 31.80s/it]
{'loss': 0.8969, 'grad_norm': 0.0733579471707344, 'learning_rate': 0.00019842575208396372, 'memory/max_active (GiB)': 28.15, 'memory/max_allocated (GiB)': 28.15, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1861.37, 'epoch': 0.46}
15%|████████████████████▊ | 66/432 [33:49<3:13:58, 31.80s/it]
16%|█████████████████████ | 67/432 [34:22<3:14:20, 31.95s/it]
{'loss': 0.8604, 'grad_norm': 0.29595091938972473, 'learning_rate': 0.00019827980707737703, 'memory/max_active (GiB)': 28.15, 'memory/max_allocated (GiB)': 28.15, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1844.94, 'epoch': 0.47}
16%|█████████████████████ | 67/432 [34:22<3:14:20, 31.95s/it]
16%|█████████████████████▍ | 68/432 [34:51<3:09:19, 31.21s/it]
{'loss': 0.9479, 'grad_norm': 0.10486430674791336, 'learning_rate': 0.00019812745200513927, 'memory/max_active (GiB)': 27.2, 'memory/max_allocated (GiB)': 27.2, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1823.2, 'epoch': 0.47}
16%|█████████████████████▍ | 68/432 [34:51<3:09:19, 31.21s/it]
16%|█████████████████████▋ | 69/432 [35:28<3:18:46, 32.85s/it]
{'loss': 0.9287, 'grad_norm': 0.13543325662612915, 'learning_rate': 0.0001979686968042461, 'memory/max_active (GiB)': 25.68, 'memory/max_allocated (GiB)': 25.68, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1821.33, 'epoch': 0.48}
16%|█████████████████████▋ | 69/432 [35:28<3:18:46, 32.85s/it]
16%|██████████████████████ | 70/432 [36:00<3:16:21, 32.55s/it]
{'loss': 0.9248, 'grad_norm': 0.07873474061489105, 'learning_rate': 0.00019780355182912626, 'memory/max_active (GiB)': 27.95, 'memory/max_allocated (GiB)': 27.95, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1771.5, 'epoch': 0.49}
16%|██████████████████████ | 70/432 [36:00<3:16:21, 32.55s/it]
16%|██████████████████████▎ | 71/432 [36:30<3:11:56, 31.90s/it]
{'loss': 0.9172, 'grad_norm': 0.06926668435335159, 'learning_rate': 0.0001976320278509663, 'memory/max_active (GiB)': 26.06, 'memory/max_allocated (GiB)': 26.06, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1875.05, 'epoch': 0.49}
16%|██████████████████████▎ | 71/432 [36:30<3:11:56, 31.90s/it]
17%|██████████████████████▋ | 72/432 [37:06<3:19:18, 33.22s/it]
{'loss': 0.8823, 'grad_norm': 0.08082268387079239, 'learning_rate': 0.0001974541360570079, 'memory/max_active (GiB)': 28.9, 'memory/max_allocated (GiB)': 28.9, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1865.89, 'epoch': 0.5}
17%|██████████████████████▋ | 72/432 [37:06<3:19:18, 33.22s/it]
17%|██████████████████████▉ | 73/432 [37:37<3:14:13, 32.46s/it]
{'loss': 0.9185, 'grad_norm': 0.07178379595279694, 'learning_rate': 0.00019726988804981844, 'memory/max_active (GiB)': 27.2, 'memory/max_allocated (GiB)': 27.2, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1854.53, 'epoch': 0.51}
17%|██████████████████████▉ | 73/432 [37:37<3:14:13, 32.46s/it]
17%|███████████████████████▎ | 74/432 [38:09<3:12:00, 32.18s/it]
{'loss': 0.9461, 'grad_norm': 0.07196955382823944, 'learning_rate': 0.00019707929584653408, 'memory/max_active (GiB)': 28.15, 'memory/max_allocated (GiB)': 28.15, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1820.12, 'epoch': 0.51}
17%|███████████████████████▎ | 74/432 [38:09<3:12:00, 32.18s/it]
17%|███████████████████████▌ | 75/432 [38:35<3:01:38, 30.53s/it]
{'loss': 1.0447, 'grad_norm': 0.07153692096471786, 'learning_rate': 0.00019688237187807594, 'memory/max_active (GiB)': 25.68, 'memory/max_allocated (GiB)': 25.68, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1809.64, 'epoch': 0.52}
17%|███████████████████████▌ | 75/432 [38:35<3:01:38, 30.53s/it]
18%|███████████████████████▉ | 76/432 [39:04<2:57:30, 29.92s/it]
{'loss': 0.8106, 'grad_norm': 0.06721945106983185, 'learning_rate': 0.00019667912898833955, 'memory/max_active (GiB)': 23.22, 'memory/max_allocated (GiB)': 23.22, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1785.2, 'epoch': 0.53}
18%|███████████████████████▉ | 76/432 [39:04<2:57:30, 29.92s/it]
18%|████████████████████████▏ | 77/432 [39:35<2:58:54, 30.24s/it]
{'loss': 0.9299, 'grad_norm': 0.08322236686944962, 'learning_rate': 0.00019646958043335677, 'memory/max_active (GiB)': 24.64, 'memory/max_allocated (GiB)': 24.64, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1836.34, 'epoch': 0.53}
18%|████████████████████████▏ | 77/432 [39:35<2:58:54, 30.24s/it]
18%|████████████████████████▌ | 78/432 [40:07<3:02:40, 30.96s/it]
{'loss': 0.9262, 'grad_norm': 0.06773433834314346, 'learning_rate': 0.00019625373988043165, 'memory/max_active (GiB)': 27.95, 'memory/max_allocated (GiB)': 27.95, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1850.19, 'epoch': 0.54}
18%|████████████████████████▌ | 78/432 [40:07<3:02:40, 30.96s/it]
18%|████████████████████████▊ | 79/432 [40:40<3:05:38, 31.55s/it]
{'loss': 0.9067, 'grad_norm': 0.06558340042829514, 'learning_rate': 0.00019603162140724862, 'memory/max_active (GiB)': 27.2, 'memory/max_allocated (GiB)': 27.2, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1876.53, 'epoch': 0.55}
18%|████████████████████████▊ | 79/432 [40:40<3:05:38, 31.55s/it]
19%|█████████████████████████▏ | 80/432 [41:07<2:56:33, 30.10s/it]
{'loss': 0.8971, 'grad_norm': 0.06962298601865768, 'learning_rate': 0.0001958032395009545, 'memory/max_active (GiB)': 25.68, 'memory/max_allocated (GiB)': 25.68, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1767.39, 'epoch': 0.56}
19%|█████████████████████████▏ | 80/432 [41:07<2:56:33, 30.10s/it]
19%|█████████████████████████▌ | 81/432 [41:32<2:47:00, 28.55s/it]
{'loss': 0.9593, 'grad_norm': 0.08821487426757812, 'learning_rate': 0.00019556860905721362, 'memory/max_active (GiB)': 25.11, 'memory/max_allocated (GiB)': 25.11, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1789.22, 'epoch': 0.56}
19%|█████████████████████████▌ | 81/432 [41:32<2:47:00, 28.55s/it]
19%|█████████████████████████▊ | 82/432 [42:00<2:45:10, 28.32s/it]
{'loss': 0.9409, 'grad_norm': 0.07249249517917633, 'learning_rate': 0.00019532774537923617, 'memory/max_active (GiB)': 27.44, 'memory/max_allocated (GiB)': 27.44, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1764.78, 'epoch': 0.57}
19%|█████████████████████████▊ | 82/432 [42:00<2:45:10, 28.32s/it]
19%|██████████████████████████▏ | 83/432 [42:29<2:46:56, 28.70s/it]
{'loss': 0.8989, 'grad_norm': 0.08999690413475037, 'learning_rate': 0.00019508066417678018, 'memory/max_active (GiB)': 28.9, 'memory/max_allocated (GiB)': 28.9, 'memory/device_reserved (GiB)': 32.39, 'tokens_per_second_per_gpu': 1821.55, 'epoch': 0.58}