初始化项目,由ModelHub XC社区提供模型

Model: nikitastheo/phonebabylm-phone-deu-eng-interleaved
Source: Original Platform
This commit is contained in:
ModelHub XC
2026-08-21 19:40:16 +08:00
commit 5cdf22bdab
11 changed files with 566 additions and 0 deletions

35
.gitattributes vendored Normal file
View File

@@ -0,0 +1,35 @@
*.7z filter=lfs diff=lfs merge=lfs -text
*.arrow filter=lfs diff=lfs merge=lfs -text
*.bin filter=lfs diff=lfs merge=lfs -text
*.bz2 filter=lfs diff=lfs merge=lfs -text
*.ckpt filter=lfs diff=lfs merge=lfs -text
*.ftz filter=lfs diff=lfs merge=lfs -text
*.gz filter=lfs diff=lfs merge=lfs -text
*.h5 filter=lfs diff=lfs merge=lfs -text
*.joblib filter=lfs diff=lfs merge=lfs -text
*.lfs.* filter=lfs diff=lfs merge=lfs -text
*.mlmodel filter=lfs diff=lfs merge=lfs -text
*.model filter=lfs diff=lfs merge=lfs -text
*.msgpack filter=lfs diff=lfs merge=lfs -text
*.npy filter=lfs diff=lfs merge=lfs -text
*.npz filter=lfs diff=lfs merge=lfs -text
*.onnx filter=lfs diff=lfs merge=lfs -text
*.ot filter=lfs diff=lfs merge=lfs -text
*.parquet filter=lfs diff=lfs merge=lfs -text
*.pb filter=lfs diff=lfs merge=lfs -text
*.pickle filter=lfs diff=lfs merge=lfs -text
*.pkl filter=lfs diff=lfs merge=lfs -text
*.pt filter=lfs diff=lfs merge=lfs -text
*.pth filter=lfs diff=lfs merge=lfs -text
*.rar filter=lfs diff=lfs merge=lfs -text
*.safetensors filter=lfs diff=lfs merge=lfs -text
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
*.tar.* filter=lfs diff=lfs merge=lfs -text
*.tar filter=lfs diff=lfs merge=lfs -text
*.tflite filter=lfs diff=lfs merge=lfs -text
*.tgz filter=lfs diff=lfs merge=lfs -text
*.wasm filter=lfs diff=lfs merge=lfs -text
*.xz filter=lfs diff=lfs merge=lfs -text
*.zip filter=lfs diff=lfs merge=lfs -text
*.zst filter=lfs diff=lfs merge=lfs -text
*tfevents* filter=lfs diff=lfs merge=lfs -text

23
README.md Normal file
View File

@@ -0,0 +1,23 @@
---
tags:
- causal-lm
- text-generation
library_name: transformers
---
# nikitastheo/phonebabylm-phone-deu-eng-interleaved
Trained with train_clm.py, a Hugging Face Accelerate causal-LM training script (no `Trainer`).
## Training details
- **Base config**: `gpt_base_config.json`
- **Tokenizer**: `nikitastheo/phonelm-phone-deu-eng-tokenizer`
- **Max steps**: 40000
- **Learning rate**: 0.0005
- **LR scheduler**: linear
- **Warmup steps**: 4000
- **Batch size (per device)**: 32
- **Gradient accumulation steps**: 1
- **Total train batch size**: 32

33
config.json Normal file
View File

@@ -0,0 +1,33 @@
{
"activation_function": "gelu",
"architectures": [
"GPT2LMHeadModel"
],
"attn_pdrop": 0.1,
"bos_token_id": 2,
"dtype": "float32",
"embd_pdrop": 0.1,
"eos_token_id": 2,
"initializer_range": 0.02,
"layer_norm_epsilon": 1e-05,
"model_type": "gpt2",
"n_ctx": 512,
"n_embd": 768,
"n_head": 12,
"n_inner": 3072,
"n_layer": 12,
"n_positions": 512,
"pad_token_id": 1,
"reorder_and_upcast_attn": false,
"resid_pdrop": 0.1,
"scale_attn_by_inverse_layer_idx": false,
"scale_attn_weights": true,
"summary_activation": null,
"summary_first_dropout": 0.1,
"summary_proj_to_labels": true,
"summary_type": "cls_index",
"summary_use_proj": true,
"transformers_version": "4.57.6",
"use_cache": true,
"vocab_size": 85
}

7
generation_config.json Normal file
View File

@@ -0,0 +1,7 @@
{
"_from_model_config": true,
"bos_token_id": 2,
"eos_token_id": 2,
"pad_token_id": 1,
"transformers_version": "4.57.6"
}

3
model.safetensors Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:12c6211529387b17fbd16b9fe0aa7fcc8f429d10b6054fac09d3b61a250af7d5
size 342072960

30
special_tokens_map.json Normal file
View File

@@ -0,0 +1,30 @@
{
"bos_token": {
"content": "UTT_BOUNDARY",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false
},
"eos_token": {
"content": "UTT_BOUNDARY",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false
},
"pad_token": {
"content": "PAD",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false
},
"unk_token": {
"content": "UNK",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false
}
}

188
tokenizer.json Normal file
View File

@@ -0,0 +1,188 @@
{
"version": "1.0",
"truncation": null,
"padding": null,
"added_tokens": [
{
"id": 0,
"content": "UNK",
"single_word": false,
"lstrip": false,
"rstrip": false,
"normalized": false,
"special": true
},
{
"id": 1,
"content": "PAD",
"single_word": false,
"lstrip": false,
"rstrip": false,
"normalized": false,
"special": true
},
{
"id": 2,
"content": "UTT_BOUNDARY",
"single_word": false,
"lstrip": false,
"rstrip": false,
"normalized": false,
"special": true
}
],
"normalizer": {
"type": "Sequence",
"normalizers": [
{
"type": "Replace",
"pattern": {
"String": "ː"
},
"content": " ː "
},
{
"type": "Strip",
"strip_left": true,
"strip_right": true
}
]
},
"pre_tokenizer": {
"type": "Whitespace"
},
"post_processor": {
"type": "TemplateProcessing",
"single": [
{
"SpecialToken": {
"id": "UTT_BOUNDARY",
"type_id": 0
}
},
{
"Sequence": {
"id": "A",
"type_id": 0
}
}
],
"pair": [
{
"Sequence": {
"id": "A",
"type_id": 0
}
},
{
"Sequence": {
"id": "B",
"type_id": 1
}
}
],
"special_tokens": {
"UTT_BOUNDARY": {
"id": "UTT_BOUNDARY",
"ids": [
2
],
"tokens": [
"UTT_BOUNDARY"
]
}
}
},
"decoder": null,
"model": {
"type": "WordLevel",
"vocab": {
"UNK": 0,
"PAD": 1,
"UTT_BOUNDARY": 2,
"ɪ": 3,
"ç": 4,
"WORD_BOUNDARY": 5,
"v": 6,
"a": 7,
"ː": 8,
"ɐ": 9,
"aʊ": 10,
"f": 11,
"ʀ": 12,
"aɪ": 13,
"z": 14,
"ə": 15,
"n": 16,
"ʊ": 17,
"t": 18,
"h": 19,
"ts": 20,
"m": 21,
"b": 22,
"ɡ": 23,
"ɔ": 24,
"d": 25,
"s": 26,
"ʃ": 27,
"i": 28,
"ʊɐ": 29,
"l": 30,
"ŋ": 31,
"ɛ": 32,
"e": 33,
"u": 34,
"k": 35,
"ʏ": 36,
"j": 37,
"p": 38,
"x": 39,
"o": 40,
"y": 41,
"œ": 42,
"pf": 43,
"ø": 44,
"w": 45,
"eɪ": 46,
"ɒ": 47,
"t̠ʃ": 48,
"d̠ʒ": 49,
"ã": 50,
"əʊ": 51,
"ɹ": 52,
"eə": 53,
"ʌ": 54,
"ɛɪ": 55,
"ð": 56,
"ɔɪ": 57,
"ʒ": 58,
"iə": 59,
"θ": 60,
"ʊə": 61,
"œ̃": 62,
"ɔ̃": 63,
"l̩": 64,
"tɕ": 65,
"ɕ": 66,
"c": 67,
"ɲ": 68,
"tʰ": 69,
"æ": 70,
"ɜ": 71,
"ɑ": 72,
"oʊ": 73,
"ʂ": 74,
"ʈ": 75,
"ɖ": 76,
"ɭ": 77,
"ʋ": 78,
"ʰχ": 79,
"ʲ": 80,
"ɟ": 81,
"ɳ": 82,
"bʰ": 83,
"dʰ": 84
},
"unk_token": "UNK"
}
}

37
tokenizer_config.json Normal file
View File

@@ -0,0 +1,37 @@
{
"add_prefix_space": false,
"added_tokens_decoder": {
"0": {
"content": "UNK",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"1": {
"content": "PAD",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"2": {
"content": "UTT_BOUNDARY",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
}
},
"bos_token": "UTT_BOUNDARY",
"clean_up_tokenization_spaces": false,
"eos_token": "UTT_BOUNDARY",
"extra_special_tokens": {},
"model_max_length": 1000000000000000019884624838656,
"pad_token": "PAD",
"tokenizer_class": "GPT2Tokenizer",
"unk_token": "UNK"
}

184
train_rank0.log Normal file
View File

@@ -0,0 +1,184 @@
07/16/2026 22:47:16 - INFO - __main__ - [RANK 0] Distributed environment: DistributedType.NO
Num processes: 1
Process index: 0
Local process index: 0
Device: cuda
Mixed precision type: no
07/16/2026 22:47:16 - INFO - __main__ - [RANK 0] Training args:
{'data_dir': 'data/phone/deu-eng/tokenized_interleaved', 'model_name_or_path': None, 'config_name': 'gpt_base_config.json', 'tokenizer_name': 'nikitastheo/phonelm-phone-deu-eng-tokenizer', 'per_device_train_batch_size': 32, 'per_device_eval_batch_size': 32, 'learning_rate': 0.0005, 'weight_decay': 0.01, 'max_steps': 40000, 'gradient_accumulation_steps': 1, 'lr_scheduler_type': <SchedulerType.LINEAR: 'linear'>, 'warmup_steps': 4000, 'output_dir': 'models/phonebabylm-phone-deu-eng-interleaved', 'seed': 42, 'model_type': None, 'block_size': None, 'push_to_hub': True, 'hub_model_id': 'nikitastheo/phonebabylm-phone-deu-eng-interleaved', 'hub_token': None, 'private': False, 'trust_remote_code': False, 'with_tracking': True, 'tracking_dir': None, 'report_to': 'wandb', 'restart_lr_at_switch': False, 'language_switch_steps': None, 'model_revision': 'main', 'config_overrides': None, 'use_slow_tokenizer': False, 'override_vocab_size': None, 'dtype': None, 'logging_steps': 10, 'cache_dir': None, 'eval_steps': 2000, 'max_grad_norm': 1.0, 'adam_epsilon': 1e-06, 'run_name': 'phonebabylm-phone-deu-eng-interleaved', 'max_train_samples': None, 'max_eval_samples': None}
07/16/2026 22:47:17 - INFO - transformers.configuration_utils - loading configuration file gpt_base_config.json
07/16/2026 22:47:17 - INFO - transformers.configuration_utils - Model config GPT2Config {
"activation_function": "gelu",
"architectures": [
"GPT2LMHeadModel"
],
"attn_pdrop": 0.1,
"bos_token_id": 50256,
"embd_pdrop": 0.1,
"eos_token_id": 50256,
"initializer_range": 0.02,
"layer_norm_epsilon": 1e-05,
"model_type": "gpt2",
"n_ctx": 512,
"n_embd": 768,
"n_head": 12,
"n_inner": 3072,
"n_layer": 12,
"n_positions": 512,
"reorder_and_upcast_attn": false,
"resid_pdrop": 0.1,
"scale_attn_by_inverse_layer_idx": false,
"scale_attn_weights": true,
"summary_activation": null,
"summary_first_dropout": 0.1,
"summary_proj_to_labels": true,
"summary_type": "cls_index",
"summary_use_proj": true,
"transformers_version": "4.57.6",
"use_cache": true,
"vocab_size": 1
}
07/16/2026 22:47:17 - INFO - transformers.tokenization_utils_base - loading file vocab.json from cache at /home/geofila/.cache/huggingface/hub/models--nikitastheo--phonelm-phone-deu-eng-tokenizer/snapshots/a82d9c161c2c55d913c3f8dae7f59fbdd01bd623/vocab.json
07/16/2026 22:47:17 - INFO - transformers.tokenization_utils_base - loading file merges.txt from cache at None
07/16/2026 22:47:17 - INFO - transformers.tokenization_utils_base - loading file tokenizer.json from cache at /home/geofila/.cache/huggingface/hub/models--nikitastheo--phonelm-phone-deu-eng-tokenizer/snapshots/a82d9c161c2c55d913c3f8dae7f59fbdd01bd623/tokenizer.json
07/16/2026 22:47:17 - INFO - transformers.tokenization_utils_base - loading file added_tokens.json from cache at None
07/16/2026 22:47:17 - INFO - transformers.tokenization_utils_base - loading file special_tokens_map.json from cache at /home/geofila/.cache/huggingface/hub/models--nikitastheo--phonelm-phone-deu-eng-tokenizer/snapshots/a82d9c161c2c55d913c3f8dae7f59fbdd01bd623/special_tokens_map.json
07/16/2026 22:47:17 - INFO - transformers.tokenization_utils_base - loading file tokenizer_config.json from cache at /home/geofila/.cache/huggingface/hub/models--nikitastheo--phonelm-phone-deu-eng-tokenizer/snapshots/a82d9c161c2c55d913c3f8dae7f59fbdd01bd623/tokenizer_config.json
07/16/2026 22:47:17 - INFO - transformers.tokenization_utils_base - loading file chat_template.jinja from cache at None
07/16/2026 22:47:19 - INFO - transformers.generation.configuration_utils - Generate config GenerationConfig {
"bos_token_id": 2,
"eos_token_id": 2,
"pad_token_id": 1
}
07/16/2026 22:47:20 - INFO - __main__ - [RANK 0] Model parameters: 85.51M total (85.51M trainable)
07/16/2026 22:47:39 - INFO - __main__ - [RANK 0] Sample 167621 of the training set: {'input_ids': tensor([ 5, 11, 30, 24, 52, 5, 56, 15, 5, 26, 30, 13, 25, 3, 31, 5, 35, 54,
22, 15, 25, 14, 5, 3, 16, 56, 15, 5, 45, 24, 30, 14, 5, 56, 15, 5,
23, 52, 46, 18, 5, 22, 28, 8, 21, 14, 5, 54, 6, 56, 15, 5, 26, 28,
8, 30, 3, 31, 5, 35, 54, 6, 15, 52, 25, 5, 45, 3, 56, 5, 19, 17,
35, 26, 5, 11, 52, 54, 21, 5, 45, 3, 48, 5, 45, 71, 8, 5, 26, 15,
26, 38, 32, 16, 25, 3, 25, 5, 11, 30, 3, 48, 3, 14, 5, 54, 6, 5,
22, 46, 35, 15, 16, 5, 22, 54, 16, 48, 3, 14, 5, 54, 6, 5, 25, 52,
13, 25, 5, 71, 8, 22, 14, 5, 26, 18, 52, 3, 31, 14, 5, 54, 6, 5,
54, 16, 59, 16, 14, 5, 70, 16, 25, 5, 28, 8, 6, 15, 16, 5, 21, 3,
26, 18, 15, 52, 5, 22, 3, 31, 35, 26, 3, 14, 5, 11, 3, 27, 3, 31,
22, 34, 8, 18, 26, 5, 24, 30, 5, 45, 71, 8, 5, 16, 34, 8, 5, 18,
15, 5, 19, 71, 8, 52, 5, 3, 16, 18, 52, 32, 26, 18, 3, 25, 5, 23,
46, 14, 5, 70, 16, 25, 5, 19, 71, 8, 5, 35, 45, 3, 35, 5, 13, 14,
5, 18, 17, 35, 5, 3, 16, 5, 32, 6, 52, 3, 60, 3, 31, 5, 11, 52,
54, 21, 56, 15, 5, 23, 54, 16, 52, 70, 35, 5, 73, 6, 15, 52, 5, 56,
15, 5, 25, 52, 32, 26, 15, 52, 5, 18, 15, 5, 56, 15, 5, 48, 13, 16,
15, 5, 25, 72, 23, 14, 5, 24, 16, 56, 15, 5, 48, 3, 21, 16, 28, 38,
28, 8, 26, 5, 56, 15, 5, 35, 3, 48, 15, 16, 5, 45, 54, 14, 5, 26,
73, 5, 30, 72, 52, 49, 5, 56, 70, 18, 5, 19, 70, 11, 5, 54, 6, 5,
3, 18, 5, 26, 28, 8, 21, 25, 5, 18, 15, 22, 28, 5, 52, 3, 14, 71,
8, 6, 25, 5, 70, 14, 5, 54, 5, 38, 72, 52, 30, 15, 52, 5, 56, 32,
52, 45, 54, 14, 5, 54, 5, 26, 35, 45, 32, 52, 5, 54, 6, 5, 35, 72,
52, 38, 3, 18, 5, 30, 46, 25, 5, 25, 10, 16, 5, 70, 18, 5, 45, 54,
16, 5, 32, 16, 25, 5, 15, 38, 72, 16, 5, 45, 3, 48, 5, 26, 18, 17,
25, 5, 54, 5, 52, 10, 16, 25, 5, 18, 46, 22, 15, 30, 5, 26, 38, 52,
32, 25, 5, 45, 3, 56, 5, 21, 3, 26, 3, 14, 5, 22, 3, 31, 35, 26,
3, 14, 5, 6, 32, 52, 28, 5, 22, 32, 26, 18, 5, 48, 13, 16, 15, 5,
18, 28, 8, 26, 71, 8, 6, 3, 26, 5, 70, 16, 25, 5, 54, 5, 26, 15,
38, 30, 13, 5, 54, 6, 5, 25])} WORD_BOUNDARY f l ɔ ɹ WORD_BOUNDARY ð ə WORD_BOUNDARY s l aɪ d ɪ ŋ WORD_BOUNDARY k ʌ b ə d z WORD_BOUNDARY ɪ n ð ə WORD_BOUNDARY w ɔ l z WORD_BOUNDARY ð ə WORD_BOUNDARY ɡ ɹ eɪ t WORD_BOUNDARY b i ː m z WORD_BOUNDARY ʌ v ð ə WORD_BOUNDARY s i ː l ɪ ŋ WORD_BOUNDARY k ʌ v ə ɹ d WORD_BOUNDARY w ɪ ð WORD_BOUNDARY h ʊ k s WORD_BOUNDARY f ɹ ʌ m WORD_BOUNDARY w ɪ t̠ʃ WORD_BOUNDARY w ɜ ː WORD_BOUNDARY s ə s p ɛ n d ɪ d WORD_BOUNDARY f l ɪ t̠ʃ ɪ z WORD_BOUNDARY ʌ v WORD_BOUNDARY b eɪ k ə n WORD_BOUNDARY b ʌ n t̠ʃ ɪ z WORD_BOUNDARY ʌ v WORD_BOUNDARY d ɹ aɪ d WORD_BOUNDARY ɜ ː b z WORD_BOUNDARY s t ɹ ɪ ŋ z WORD_BOUNDARY ʌ v WORD_BOUNDARY ʌ n iə n z WORD_BOUNDARY æ n d WORD_BOUNDARY i ː v ə n WORD_BOUNDARY m ɪ s t ə ɹ WORD_BOUNDARY b ɪ ŋ k s ɪ z WORD_BOUNDARY f ɪ ʃ ɪ ŋ b u ː t s WORD_BOUNDARY ɔ l WORD_BOUNDARY w ɜ ː WORD_BOUNDARY n u ː WORD_BOUNDARY t ə WORD_BOUNDARY h ɜ ː ɹ WORD_BOUNDARY ɪ n t ɹ ɛ s t ɪ d WORD_BOUNDARY ɡ eɪ z WORD_BOUNDARY æ n d WORD_BOUNDARY h ɜ ː WORD_BOUNDARY k w ɪ k WORD_BOUNDARY aɪ z WORD_BOUNDARY t ʊ k WORD_BOUNDARY ɪ n WORD_BOUNDARY ɛ v ɹ ɪ θ ɪ ŋ WORD_BOUNDARY f ɹ ʌ m ð ə WORD_BOUNDARY ɡ ʌ n ɹ æ k WORD_BOUNDARY oʊ v ə ɹ WORD_BOUNDARY ð ə WORD_BOUNDARY d ɹ ɛ s ə ɹ WORD_BOUNDARY t ə WORD_BOUNDARY ð ə WORD_BOUNDARY t̠ʃ aɪ n ə WORD_BOUNDARY d ɑ ɡ z WORD_BOUNDARY ɔ n ð ə WORD_BOUNDARY t̠ʃ ɪ m n i p i ː s WORD_BOUNDARY ð ə WORD_BOUNDARY k ɪ t̠ʃ ə n WORD_BOUNDARY w ʌ z WORD_BOUNDARY s oʊ WORD_BOUNDARY l ɑ ɹ d̠ʒ WORD_BOUNDARY ð æ t WORD_BOUNDARY h æ f WORD_BOUNDARY ʌ v WORD_BOUNDARY ɪ t WORD_BOUNDARY s i ː m d WORD_BOUNDARY t ə b i WORD_BOUNDARY ɹ ɪ z ɜ ː v d WORD_BOUNDARY æ z WORD_BOUNDARY ʌ WORD_BOUNDARY p ɑ ɹ l ə ɹ WORD_BOUNDARY ð ɛ ɹ w ʌ z WORD_BOUNDARY ʌ WORD_BOUNDARY s k w ɛ ɹ WORD_BOUNDARY ʌ v WORD_BOUNDARY k ɑ ɹ p ɪ t WORD_BOUNDARY l eɪ d WORD_BOUNDARY d aʊ n WORD_BOUNDARY æ t WORD_BOUNDARY w ʌ n WORD_BOUNDARY ɛ n d WORD_BOUNDARY ə p ɑ n WORD_BOUNDARY w ɪ t̠ʃ WORD_BOUNDARY s t ʊ d WORD_BOUNDARY ʌ WORD_BOUNDARY ɹ aʊ n d WORD_BOUNDARY t eɪ b ə l WORD_BOUNDARY s p ɹ ɛ d WORD_BOUNDARY w ɪ ð WORD_BOUNDARY m ɪ s ɪ z WORD_BOUNDARY b ɪ ŋ k s ɪ z WORD_BOUNDARY v ɛ ɹ i WORD_BOUNDARY b ɛ s t WORD_BOUNDARY t̠ʃ aɪ n ə WORD_BOUNDARY t i ː s ɜ ː v ɪ s WORD_BOUNDARY æ n d WORD_BOUNDARY ʌ WORD_BOUNDARY s ə p l aɪ WORD_BOUNDARY ʌ v WORD_BOUNDARY d
07/16/2026 22:47:39 - INFO - __main__ - [RANK 0] Sample 29184 of the training set: {'input_ids': tensor([13, 16, 15, 5, 18, 7, 27, 15, 16, 5, 21, 3, 18, 5, 35, 17, 38, 11,
9, 23, 32, 30, 18, 5, 11, 32, 9, 27, 30, 40, 8, 26, 5, 25, 28, 8,
5, 35, 3, 26, 18, 15, 5, 14, 32, 20, 18, 15, 5, 25, 33, 8, 16, 5,
19, 17, 16, 18, 5, 6, 28, 8, 25, 9, 5, 19, 3, 16, 10, 11, 5, 17,
16, 18, 5, 23, 3, 31, 5, 3, 16, 5, 25, 7, 26, 5, 7, 16, 25, 15,
12, 15, 5, 20, 3, 21, 9, 5, 38, 24, 20, 18, 10, 14, 15, 16, 18, 5,
25, 7, 8, 5, 14, 7, 8, 26, 5, 25, 32, 9, 5, 19, 17, 16, 18, 5,
21, 3, 18, 5, 10, 23, 15, 16, 5, 14, 40, 8, 5, 23, 9, 40, 8, 26,
5, 6, 28, 8, 5, 21, 41, 8, 30, 12, 32, 8, 25, 9, 5, 2, 23, 30,
24, 20, 5, 21, 3, 4, 5, 16, 3, 4, 18, 5, 14, 40, 8, 5, 7, 16,
5, 14, 7, 8, 23, 18, 15, 5, 25, 32, 9, 5, 14, 24, 30, 25, 7, 8,
18, 5, 17, 16, 18, 5, 14, 32, 20, 18, 15, 5, 25, 33, 8, 16, 5, 19,
17, 16, 18, 5, 10, 11, 5, 25, 28, 8, 5, 27, 36, 9, 20, 15, 5, 7,
30, 26, 5, 32, 9, 5, 7, 8, 22, 9, 5, 25, 7, 26, 5, 11, 28, 8,
30, 15, 5, 14, 3, 30, 22, 9, 23, 32, 30, 18, 5, 14, 7, 8, 5, 6,
7, 9, 11, 5, 32, 9, 5, 7, 30, 15, 26, 5, 35, 17, 38, 11, 9, 23,
32, 30, 18, 5, 11, 24, 9, 18, 5, 17, 16, 18, 5, 11, 36, 30, 18, 15,
5, 14, 3, 4, 5, 25, 28, 8, 5, 18, 7, 27, 15, 16, 5, 17, 16, 18,
5, 25, 33, 8, 16, 5, 18, 24, 9, 16, 3, 26, 18, 9, 5, 21, 3, 18,
5, 14, 3, 30, 22, 9, 5, 25, 7, 16, 5, 23, 3, 31, 5, 32, 9, 5,
3, 16, 5, 25, 28, 8, 5, 25, 9, 3, 18, 15, 5, 35, 7, 21, 9, 5,
6, 40, 8, 5, 25, 32, 9, 5, 19, 17, 16, 18, 5, 6, 7, 8, 9, 5,
21, 3, 18, 5, 10, 23, 15, 16, 5, 14, 40, 8, 5, 23, 9, 40, 8, 26,
5, 6, 28, 8, 5, 13, 16, 5, 12, 17, 16, 25, 9, 5, 18, 29, 21, 5,
2, 23, 34, 8, 18, 15, 16, 5, 7, 8, 22, 15, 16, 18, 5, 14, 7, 8,
23, 18, 15, 5, 25, 32, 9, 5, 14, 24, 30, 25, 7, 8, 18, 5, 19, 40,
8, 38, 5, 25, 33, 8, 16, 5, 19, 17, 16, 18, 5, 19, 32, 12, 17, 16,
18, 9, 5, 17, 16, 18, 5, 42, 11, 16, 15, 18, 15, 5, 25, 28, 8, 5,
35, 3, 26, 18, 15, 5, 6, 7])} aɪ n ə WORD_BOUNDARY t a ʃ ə n WORD_BOUNDARY m ɪ t WORD_BOUNDARY k ʊ p f ɐ ɡ ɛ l t WORD_BOUNDARY f ɛ ɐ ʃ l o ː s WORD_BOUNDARY d i ː WORD_BOUNDARY k ɪ s t ə WORD_BOUNDARY z ɛ ts t ə WORD_BOUNDARY d e ː n WORD_BOUNDARY h ʊ n t WORD_BOUNDARY v i ː d ɐ WORD_BOUNDARY h ɪ n aʊ f WORD_BOUNDARY ʊ n t WORD_BOUNDARY ɡ ɪ ŋ WORD_BOUNDARY ɪ n WORD_BOUNDARY d a s WORD_BOUNDARY a n d ə ʀ ə WORD_BOUNDARY ts ɪ m ɐ WORD_BOUNDARY p ɔ ts t aʊ z ə n t WORD_BOUNDARY d a ː WORD_BOUNDARY z a ː s WORD_BOUNDARY d ɛ ɐ WORD_BOUNDARY h ʊ n t WORD_BOUNDARY m ɪ t WORD_BOUNDARY aʊ ɡ ə n WORD_BOUNDARY z o ː WORD_BOUNDARY ɡ ɐ o ː s WORD_BOUNDARY v i ː WORD_BOUNDARY m y ː l ʀ ɛ ː d ɐ WORD_BOUNDARY UTT_BOUNDARY ɡ l ɔ ts WORD_BOUNDARY m ɪ ç WORD_BOUNDARY n ɪ ç t WORD_BOUNDARY z o ː WORD_BOUNDARY a n WORD_BOUNDARY z a ː ɡ t ə WORD_BOUNDARY d ɛ ɐ WORD_BOUNDARY z ɔ l d a ː t WORD_BOUNDARY ʊ n t WORD_BOUNDARY z ɛ ts t ə WORD_BOUNDARY d e ː n WORD_BOUNDARY h ʊ n t WORD_BOUNDARY aʊ f WORD_BOUNDARY d i ː WORD_BOUNDARY ʃ ʏ ɐ ts ə WORD_BOUNDARY a l s WORD_BOUNDARY ɛ ɐ WORD_BOUNDARY a ː b ɐ WORD_BOUNDARY d a s WORD_BOUNDARY f i ː l ə WORD_BOUNDARY z ɪ l b ɐ ɡ ɛ l t WORD_BOUNDARY z a ː WORD_BOUNDARY v a ɐ f WORD_BOUNDARY ɛ ɐ WORD_BOUNDARY a l ə s WORD_BOUNDARY k ʊ p f ɐ ɡ ɛ l t WORD_BOUNDARY f ɔ ɐ t WORD_BOUNDARY ʊ n t WORD_BOUNDARY f ʏ l t ə WORD_BOUNDARY z ɪ ç WORD_BOUNDARY d i ː WORD_BOUNDARY t a ʃ ə n WORD_BOUNDARY ʊ n t WORD_BOUNDARY d e ː n WORD_BOUNDARY t ɔ ɐ n ɪ s t ɐ WORD_BOUNDARY m ɪ t WORD_BOUNDARY z ɪ l b ɐ WORD_BOUNDARY d a n WORD_BOUNDARY ɡ ɪ ŋ WORD_BOUNDARY ɛ ɐ WORD_BOUNDARY ɪ n WORD_BOUNDARY d i ː WORD_BOUNDARY d ɐ ɪ t ə WORD_BOUNDARY k a m ɐ WORD_BOUNDARY v o ː WORD_BOUNDARY d ɛ ɐ WORD_BOUNDARY h ʊ n t WORD_BOUNDARY v a ː ɐ WORD_BOUNDARY m ɪ t WORD_BOUNDARY aʊ ɡ ə n WORD_BOUNDARY z o ː WORD_BOUNDARY ɡ ɐ o ː s WORD_BOUNDARY v i ː WORD_BOUNDARY aɪ n WORD_BOUNDARY ʀ ʊ n d ɐ WORD_BOUNDARY t ʊɐ m WORD_BOUNDARY UTT_BOUNDARY ɡ u ː t ə n WORD_BOUNDARY a ː b ə n t WORD_BOUNDARY z a ː ɡ t ə WORD_BOUNDARY d ɛ ɐ WORD_BOUNDARY z ɔ l d a ː t WORD_BOUNDARY h o ː p WORD_BOUNDARY d e ː n WORD_BOUNDARY h ʊ n t WORD_BOUNDARY h ɛ ʀ ʊ n t ɐ WORD_BOUNDARY ʊ n t WORD_BOUNDARY œ f n ə t ə WORD_BOUNDARY d i ː WORD_BOUNDARY k ɪ s t ə WORD_BOUNDARY v a
07/16/2026 22:47:39 - INFO - __main__ - [RANK 0] Sample 6556 of the training set: {'input_ids': tensor([16, 5, 2, 6, 32, 8, 12, 15, 16, 18, 5, 25, 32, 9, 5, 27, 34, 8,
30, 11, 33, 8, 12, 28, 8, 15, 16, 5, 11, 28, 8, 30, 5, 10, 39, 5,
25, 28, 8, 5, 14, 24, 16, 18, 7, 35, 26, 4, 34, 8, 30, 15, 5, 10,
26, 5, 18, 9, 24, 20, 25, 33, 8, 21, 5, 6, 7, 8, 9, 5, 19, 24,
36, 18, 5, 7, 30, 15, 26, 5, 11, 12, 41, 8, 20, 13, 18, 3, 4, 5,
3, 16, 5, 25, 32, 9, 5, 35, 3, 9, 4, 15, 5, 25, 7, 26, 5, 10,
11, 12, 33, 8, 23, 15, 16, 25, 15, 5, 32, 9, 13, 23, 16, 3, 26, 5,
6, 29, 25, 15, 5, 30, 33, 8, 38, 19, 7, 11, 18, 5, 32, 9, 42, 9,
18, 9, 18, 5, 32, 26, 5, 6, 29, 25, 15, 5, 32, 9, 20, 32, 8, 30,
18, 5, 25, 7, 26, 5, 16, 24, 39, 5, 35, 13, 16, 15, 5, 27, 38, 34,
8, 9, 5, 11, 24, 16, 5, 25, 33, 8, 16, 5, 30, 7, 16, 18, 27, 18,
9, 13, 4, 9, 16, 5, 23, 15, 11, 17, 16, 25, 15, 16, 5, 6, 24, 9,
25, 15, 16, 5, 14, 13, 5, 7, 30, 26, 5, 25, 28, 8, 5, 38, 9, 33,
8, 25, 3, 4, 18, 5, 20, 34, 8, 5, 32, 16, 25, 15, 5, 6, 7, 8,
9, 5, 23, 3, 31, 5, 25, 28, 8, 5, 11, 12, 10, 5, 25, 32, 26, 5,
12, 3, 4, 18, 9, 26, 5, 18, 7, 48, 9, 5, 10, 11, 5, 11, 12, 10,
5, 19, 7, 9, 38, 9, 5, 20, 34, 8, 5, 25, 28, 8, 5, 21, 3, 18,
5, 25, 32, 9, 5, 23, 9, 40, 8, 26, 15, 16, 5, 21, 32, 31, 15, 5,
25, 33, 8, 16, 5, 23, 7, 31, 5, 19, 3, 16, 17, 16, 18, 9, 27, 12,
3, 18, 5, 17, 16, 18, 5, 14, 7, 8, 23, 18, 15, 5, 6, 3, 30, 5,
21, 13, 16, 15, 5, 22, 32, 35, 28, 8, 5, 25, 32, 16, 5, 25, 33, 8,
16, 5, 23, 7, 16, 20, 15, 16, 5, 18, 7, 8, 35, 5, 27, 30, 7, 8,
11, 15, 16, 5, 19, 7, 8, 38, 26, 5, 21, 28, 8, 9, 5, 7, 8, 22,
9, 5, 6, 40, 8, 30, 5, 23, 15, 25, 7, 39, 18, 5, 25, 7, 26, 5,
14, 28, 8, 5, 18, 24, 25, 21, 41, 8, 25, 15, 5, 14, 13, 16, 5, 6,
36, 9, 25, 15, 5, 2, 28, 8, 12, 15, 5, 22, 32, 35, 28, 8, 5, 2,
11, 12, 13, 30, 3, 4, 5, 21, 3, 18, 5, 32, 9, 27, 12, 24, 35, 15,
16, 15, 21, 5, 22, 30, 3, 35, 5, 22, 30, 28, 8, 38, 5, 14, 28, 8,
5, 25, 32, 16, 5, 7, 8, 22])} n WORD_BOUNDARY UTT_BOUNDARY v ɛ ː ʀ ə n t WORD_BOUNDARY d ɛ ɐ WORD_BOUNDARY ʃ u ː l f e ː ʀ i ː ə n WORD_BOUNDARY f i ː l WORD_BOUNDARY aʊ x WORD_BOUNDARY d i ː WORD_BOUNDARY z ɔ n t a k s ç u ː l ə WORD_BOUNDARY aʊ s WORD_BOUNDARY t ɐ ɔ ts d e ː m WORD_BOUNDARY v a ː ɐ WORD_BOUNDARY h ɔ ʏ t WORD_BOUNDARY a l ə s WORD_BOUNDARY f ʀ y ː ts aɪ t ɪ ç WORD_BOUNDARY ɪ n WORD_BOUNDARY d ɛ ɐ WORD_BOUNDARY k ɪ ɐ ç ə WORD_BOUNDARY d a s WORD_BOUNDARY aʊ f ʀ e ː ɡ ə n d ə WORD_BOUNDARY ɛ ɐ aɪ ɡ n ɪ s WORD_BOUNDARY v ʊɐ d ə WORD_BOUNDARY l e ː p h a f t WORD_BOUNDARY ɛ ɐ œ ɐ t ɐ t WORD_BOUNDARY ɛ s WORD_BOUNDARY v ʊɐ d ə WORD_BOUNDARY ɛ ɐ ts ɛ ː l t WORD_BOUNDARY d a s WORD_BOUNDARY n ɔ x WORD_BOUNDARY k aɪ n ə WORD_BOUNDARY ʃ p u ː ɐ WORD_BOUNDARY f ɔ n WORD_BOUNDARY d e ː n WORD_BOUNDARY l a n t ʃ t ɐ aɪ ç ɐ n WORD_BOUNDARY ɡ ə f ʊ n d ə n WORD_BOUNDARY v ɔ ɐ d ə n WORD_BOUNDARY z aɪ WORD_BOUNDARY a l s WORD_BOUNDARY d i ː WORD_BOUNDARY p ɐ e ː d ɪ ç t WORD_BOUNDARY ts u ː WORD_BOUNDARY ɛ n d ə WORD_BOUNDARY v a ː ɐ WORD_BOUNDARY ɡ ɪ ŋ WORD_BOUNDARY d i ː WORD_BOUNDARY f ʀ aʊ WORD_BOUNDARY d ɛ s WORD_BOUNDARY ʀ ɪ ç t ɐ s WORD_BOUNDARY t a t̠ʃ ɐ WORD_BOUNDARY aʊ f WORD_BOUNDARY f ʀ aʊ WORD_BOUNDARY h a ɐ p ɐ WORD_BOUNDARY ts u ː WORD_BOUNDARY d i ː WORD_BOUNDARY m ɪ t WORD_BOUNDARY d ɛ ɐ WORD_BOUNDARY ɡ ɐ o ː s ə n WORD_BOUNDARY m ɛ ŋ ə WORD_BOUNDARY d e ː n WORD_BOUNDARY ɡ a ŋ WORD_BOUNDARY h ɪ n ʊ n t ɐ ʃ ʀ ɪ t WORD_BOUNDARY ʊ n t WORD_BOUNDARY z a ː ɡ t ə WORD_BOUNDARY v ɪ l WORD_BOUNDARY m aɪ n ə WORD_BOUNDARY b ɛ k i ː WORD_BOUNDARY d ɛ n WORD_BOUNDARY d e ː n WORD_BOUNDARY ɡ a n ts ə n WORD_BOUNDARY t a ː k WORD_BOUNDARY ʃ l a ː f ə n WORD_BOUNDARY h a ː p s WORD_BOUNDARY m i ː ɐ WORD_BOUNDARY a ː b ɐ WORD_BOUNDARY v o ː l WORD_BOUNDARY ɡ ə d a x t WORD_BOUNDARY d a s WORD_BOUNDARY z i ː WORD_BOUNDARY t ɔ d m y ː d ə WORD_BOUNDARY z aɪ n WORD_BOUNDARY v ʏ ɐ d ə WORD_BOUNDARY UTT_BOUNDARY i ː ʀ ə WORD_BOUNDARY b ɛ k i ː WORD_BOUNDARY UTT_BOUNDARY f ʀ aɪ l ɪ ç WORD_BOUNDARY m ɪ t WORD_BOUNDARY ɛ ɐ ʃ ʀ ɔ k ə n ə m WORD_BOUNDARY b l ɪ k WORD_BOUNDARY b l i ː p WORD_BOUNDARY z i ː WORD_BOUNDARY d ɛ n WORD_BOUNDARY a ː b
07/16/2026 22:47:42 - INFO - __main__ - [RANK 0] ***** Running training *****
07/16/2026 22:47:42 - INFO - __main__ - [RANK 0] Num examples = 203590
07/16/2026 22:47:42 - INFO - __main__ - [RANK 0] Num Training Steps = 40000
07/16/2026 22:47:42 - INFO - __main__ - [RANK 0] Instantaneous batch size per device = 32
07/16/2026 22:47:42 - INFO - __main__ - [RANK 0] Total train batch size (w. parallel, distributed & accumulation) = 32
07/16/2026 22:47:42 - INFO - __main__ - [RANK 0] Gradient Accumulation steps = 1
07/16/2026 22:47:42 - INFO - __main__ - [RANK 0] Total optimization steps = 40000
07/16/2026 22:47:42 - INFO - transformers.configuration_utils - Configuration saved in models/phonebabylm-phone-deu-eng-interleaved/config.json
07/16/2026 22:47:42 - INFO - transformers.generation.configuration_utils - Configuration saved in models/phonebabylm-phone-deu-eng-interleaved/generation_config.json
07/16/2026 22:47:45 - INFO - transformers.modeling_utils - Model weights saved in models/phonebabylm-phone-deu-eng-interleaved/model.safetensors
07/16/2026 22:47:45 - INFO - transformers.tokenization_utils_base - tokenizer config file saved in models/phonebabylm-phone-deu-eng-interleaved/tokenizer_config.json
07/16/2026 22:47:45 - INFO - transformers.tokenization_utils_base - Special tokens file saved in models/phonebabylm-phone-deu-eng-interleaved/special_tokens_map.json
07/16/2026 22:47:50 - INFO - __main__ - [RANK 0] Starting epoch 0...
07/16/2026 22:47:51 - WARNING - transformers.modeling_utils - `loss_type=None` was set in the config but it is unrecognized. Using the default loss: `ForCausalLMLoss`.
07/16/2026 22:58:50 - INFO - __main__ - [RANK 0] perplexity: 3.4981748920090556 eval_loss: 1.2522413730621338 accuracy: 0.6286
07/16/2026 23:09:55 - INFO - __main__ - [RANK 0] perplexity: 3.090224323166524 eval_loss: 1.1282436847686768 accuracy: 0.6635
07/16/2026 23:20:59 - INFO - __main__ - [RANK 0] perplexity: 2.8887235844293055 eval_loss: 1.0608147382736206 accuracy: 0.6829
07/16/2026 23:22:58 - INFO - __main__ - [RANK 0] Starting epoch 1...
07/16/2026 23:32:04 - INFO - __main__ - [RANK 0] perplexity: 2.7866070328443007 eval_loss: 1.0248247385025024 accuracy: 0.6932
07/16/2026 23:43:09 - INFO - __main__ - [RANK 0] perplexity: 2.717227754278576 eval_loss: 0.9996121525764465 accuracy: 0.7002
07/16/2026 23:54:14 - INFO - __main__ - [RANK 0] perplexity: 2.662045793634297 eval_loss: 0.979094922542572 accuracy: 0.7061
07/16/2026 23:58:13 - INFO - __main__ - [RANK 0] Starting epoch 2...
07/17/2026 00:05:19 - INFO - __main__ - [RANK 0] perplexity: 2.6255481989944776 eval_loss: 0.9652897119522095 accuracy: 0.7102
07/17/2026 00:16:23 - INFO - __main__ - [RANK 0] perplexity: 2.5913564344248097 eval_loss: 0.9521814584732056 accuracy: 0.7138
07/17/2026 00:27:28 - INFO - __main__ - [RANK 0] perplexity: 2.5649778532126026 eval_loss: 0.9419498443603516 accuracy: 0.7169
07/17/2026 00:33:26 - INFO - __main__ - [RANK 0] Starting epoch 3...
07/17/2026 00:38:43 - INFO - __main__ - [RANK 0] perplexity: 2.5450176439772125 eval_loss: 0.9341375827789307 accuracy: 0.7196
07/17/2026 00:50:44 - INFO - __main__ - [RANK 0] perplexity: 2.523804941363755 eval_loss: 0.9257676601409912 accuracy: 0.7216
07/17/2026 01:02:02 - INFO - __main__ - [RANK 0] perplexity: 2.504542042072227 eval_loss: 0.9181059002876282 accuracy: 0.7241
07/17/2026 01:10:00 - INFO - __main__ - [RANK 0] Starting epoch 4...
07/17/2026 01:13:06 - INFO - __main__ - [RANK 0] perplexity: 2.4924423233459936 eval_loss: 0.9132630825042725 accuracy: 0.7257
07/17/2026 01:24:11 - INFO - __main__ - [RANK 0] perplexity: 2.4772802757896057 eval_loss: 0.907161295413971 accuracy: 0.7277
07/17/2026 01:35:16 - INFO - __main__ - [RANK 0] perplexity: 2.4638050764913904 eval_loss: 0.9017069339752197 accuracy: 0.7294
07/17/2026 01:45:13 - INFO - __main__ - [RANK 0] Starting epoch 5...
07/17/2026 01:46:20 - INFO - __main__ - [RANK 0] perplexity: 2.4591889317422853 eval_loss: 0.8998315930366516 accuracy: 0.7302
07/17/2026 01:57:25 - INFO - __main__ - [RANK 0] perplexity: 2.44883489662295 eval_loss: 0.895612359046936 accuracy: 0.7316
07/17/2026 02:08:31 - INFO - __main__ - [RANK 0] perplexity: 2.4370661854301496 eval_loss: 0.8907949328422546 accuracy: 0.7328
07/17/2026 02:19:36 - INFO - __main__ - [RANK 0] perplexity: 2.4286724169445133 eval_loss: 0.8873447775840759 accuracy: 0.7338
07/17/2026 02:20:35 - INFO - __main__ - [RANK 0] Starting epoch 6...
07/17/2026 02:30:41 - INFO - __main__ - [RANK 0] perplexity: 2.432660355583435 eval_loss: 0.8889854550361633 accuracy: 0.7341

25
training_metadata.json Normal file
View File

@@ -0,0 +1,25 @@
{
"input_config": {
"model_type": "gpt2",
"architectures": [
"GPT2LMHeadModel"
],
"activation_function": "gelu",
"attn_pdrop": 0.1,
"embd_pdrop": 0.1,
"resid_pdrop": 0.1,
"initializer_range": 0.02,
"layer_norm_epsilon": 1e-05,
"n_embd": 768,
"n_inner": 3072,
"n_head": 12,
"n_layer": 12,
"vocab_size": 1,
"n_ctx": 512,
"n_positions": 512
},
"num_steps_switch": null,
"save_steps": [
40000
]
}

1
vocab.json Normal file
View File

@@ -0,0 +1 @@
{"UNK":0,"PAD":1,"UTT_BOUNDARY":2,"ɪ":3,"ç":4,"WORD_BOUNDARY":5,"v":6,"a":7,"ː":8,"ɐ":9,"aʊ":10,"f":11,"ʀ":12,"aɪ":13,"z":14,"ə":15,"n":16,"ʊ":17,"t":18,"h":19,"ts":20,"m":21,"b":22,"ɡ":23,"ɔ":24,"d":25,"s":26,"ʃ":27,"i":28,"ʊɐ":29,"l":30,"ŋ":31,"ɛ":32,"e":33,"u":34,"k":35,"ʏ":36,"j":37,"p":38,"x":39,"o":40,"y":41,"œ":42,"pf":43,"ø":44,"w":45,"eɪ":46,"ɒ":47,"t̠ʃ":48,"d̠ʒ":49,"ã":50,"əʊ":51,"ɹ":52,"eə":53,"ʌ":54,"ɛɪ":55,"ð":56,"ɔɪ":57,"ʒ":58,"iə":59,"θ":60,"ʊə":61,"œ̃":62,"ɔ̃":63,"l̩":64,"tɕ":65,"ɕ":66,"c":67,"ɲ":68,"tʰ":69,"æ":70,"ɜ":71,"ɑ":72,"oʊ":73,"ʂ":74,"ʈ":75,"ɖ":76,"ɭ":77,"ʋ":78,"ʰχ":79,"ʲ":80,"ɟ":81,"ɳ":82,"bʰ":83,"dʰ":84}