From 5cdf22bdabca113b4a1ed0d6c7fca4794fdf7639 Mon Sep 17 00:00:00 2001 From: ModelHub XC Date: Fri, 21 Aug 2026 19:40:16 +0800 Subject: [PATCH] =?UTF-8?q?=E5=88=9D=E5=A7=8B=E5=8C=96=E9=A1=B9=E7=9B=AE?= =?UTF-8?q?=EF=BC=8C=E7=94=B1ModelHub=20XC=E7=A4=BE=E5=8C=BA=E6=8F=90?= =?UTF-8?q?=E4=BE=9B=E6=A8=A1=E5=9E=8B?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Model: nikitastheo/phonebabylm-phone-deu-eng-interleaved Source: Original Platform --- .gitattributes | 35 ++++++++ README.md | 23 +++++ config.json | 33 +++++++ generation_config.json | 7 ++ model.safetensors | 3 + special_tokens_map.json | 30 +++++++ tokenizer.json | 188 ++++++++++++++++++++++++++++++++++++++++ tokenizer_config.json | 37 ++++++++ train_rank0.log | 184 +++++++++++++++++++++++++++++++++++++++ training_metadata.json | 25 ++++++ vocab.json | 1 + 11 files changed, 566 insertions(+) create mode 100644 .gitattributes create mode 100644 README.md create mode 100644 config.json create mode 100644 generation_config.json create mode 100644 model.safetensors create mode 100644 special_tokens_map.json create mode 100644 tokenizer.json create mode 100644 tokenizer_config.json create mode 100644 train_rank0.log create mode 100644 training_metadata.json create mode 100644 vocab.json diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..a6344aa --- /dev/null +++ b/.gitattributes @@ -0,0 +1,35 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text diff --git a/README.md b/README.md new file mode 100644 index 0000000..ac836be --- /dev/null +++ b/README.md @@ -0,0 +1,23 @@ +--- +tags: +- causal-lm +- text-generation +library_name: transformers +--- + +# nikitastheo/phonebabylm-phone-deu-eng-interleaved + +Trained with train_clm.py, a Hugging Face Accelerate causal-LM training script (no `Trainer`). + +## Training details + +- **Base config**: `gpt_base_config.json` +- **Tokenizer**: `nikitastheo/phonelm-phone-deu-eng-tokenizer` +- **Max steps**: 40000 +- **Learning rate**: 0.0005 +- **LR scheduler**: linear +- **Warmup steps**: 4000 +- **Batch size (per device)**: 32 +- **Gradient accumulation steps**: 1 +- **Total train batch size**: 32 + diff --git a/config.json b/config.json new file mode 100644 index 0000000..ed20852 --- /dev/null +++ b/config.json @@ -0,0 +1,33 @@ +{ + "activation_function": "gelu", + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.1, + "bos_token_id": 2, + "dtype": "float32", + "embd_pdrop": 0.1, + "eos_token_id": 2, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_ctx": 512, + "n_embd": 768, + "n_head": 12, + "n_inner": 3072, + "n_layer": 12, + "n_positions": 512, + "pad_token_id": 1, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.1, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "transformers_version": "4.57.6", + "use_cache": true, + "vocab_size": 85 +} diff --git a/generation_config.json b/generation_config.json new file mode 100644 index 0000000..32a3d06 --- /dev/null +++ b/generation_config.json @@ -0,0 +1,7 @@ +{ + "_from_model_config": true, + "bos_token_id": 2, + "eos_token_id": 2, + "pad_token_id": 1, + "transformers_version": "4.57.6" +} diff --git a/model.safetensors b/model.safetensors new file mode 100644 index 0000000..b267a15 --- /dev/null +++ b/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:12c6211529387b17fbd16b9fe0aa7fcc8f429d10b6054fac09d3b61a250af7d5 +size 342072960 diff --git a/special_tokens_map.json b/special_tokens_map.json new file mode 100644 index 0000000..cdd0a22 --- /dev/null +++ b/special_tokens_map.json @@ -0,0 +1,30 @@ +{ + "bos_token": { + "content": "UTT_BOUNDARY", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false + }, + "eos_token": { + "content": "UTT_BOUNDARY", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false + }, + "pad_token": { + "content": "PAD", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false + }, + "unk_token": { + "content": "UNK", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false + } +} diff --git a/tokenizer.json b/tokenizer.json new file mode 100644 index 0000000..7a556d2 --- /dev/null +++ b/tokenizer.json @@ -0,0 +1,188 @@ +{ + "version": "1.0", + "truncation": null, + "padding": null, + "added_tokens": [ + { + "id": 0, + "content": "UNK", + "single_word": false, + "lstrip": false, + "rstrip": false, + "normalized": false, + "special": true + }, + { + "id": 1, + "content": "PAD", + "single_word": false, + "lstrip": false, + "rstrip": false, + "normalized": false, + "special": true + }, + { + "id": 2, + "content": "UTT_BOUNDARY", + "single_word": false, + "lstrip": false, + "rstrip": false, + "normalized": false, + "special": true + } + ], + "normalizer": { + "type": "Sequence", + "normalizers": [ + { + "type": "Replace", + "pattern": { + "String": "ː" + }, + "content": " ː " + }, + { + "type": "Strip", + "strip_left": true, + "strip_right": true + } + ] + }, + "pre_tokenizer": { + "type": "Whitespace" + }, + "post_processor": { + "type": "TemplateProcessing", + "single": [ + { + "SpecialToken": { + "id": "UTT_BOUNDARY", + "type_id": 0 + } + }, + { + "Sequence": { + "id": "A", + "type_id": 0 + } + } + ], + "pair": [ + { + "Sequence": { + "id": "A", + "type_id": 0 + } + }, + { + "Sequence": { + "id": "B", + "type_id": 1 + } + } + ], + "special_tokens": { + "UTT_BOUNDARY": { + "id": "UTT_BOUNDARY", + "ids": [ + 2 + ], + "tokens": [ + "UTT_BOUNDARY" + ] + } + } + }, + "decoder": null, + "model": { + "type": "WordLevel", + "vocab": { + "UNK": 0, + "PAD": 1, + "UTT_BOUNDARY": 2, + "ɪ": 3, + "ç": 4, + "WORD_BOUNDARY": 5, + "v": 6, + "a": 7, + "ː": 8, + "ɐ": 9, + "aʊ": 10, + "f": 11, + "ʀ": 12, + "aɪ": 13, + "z": 14, + "ə": 15, + "n": 16, + "ʊ": 17, + "t": 18, + "h": 19, + "ts": 20, + "m": 21, + "b": 22, + "ɡ": 23, + "ɔ": 24, + "d": 25, + "s": 26, + "ʃ": 27, + "i": 28, + "ʊɐ": 29, + "l": 30, + "ŋ": 31, + "ɛ": 32, + "e": 33, + "u": 34, + "k": 35, + "ʏ": 36, + "j": 37, + "p": 38, + "x": 39, + "o": 40, + "y": 41, + "œ": 42, + "pf": 43, + "ø": 44, + "w": 45, + "eɪ": 46, + "ɒ": 47, + "t̠ʃ": 48, + "d̠ʒ": 49, + "ã": 50, + "əʊ": 51, + "ɹ": 52, + "eə": 53, + "ʌ": 54, + "ɛɪ": 55, + "ð": 56, + "ɔɪ": 57, + "ʒ": 58, + "iə": 59, + "θ": 60, + "ʊə": 61, + "œ̃": 62, + "ɔ̃": 63, + "l̩": 64, + "tɕ": 65, + "ɕ": 66, + "c": 67, + "ɲ": 68, + "tʰ": 69, + "æ": 70, + "ɜ": 71, + "ɑ": 72, + "oʊ": 73, + "ʂ": 74, + "ʈ": 75, + "ɖ": 76, + "ɭ": 77, + "ʋ": 78, + "ʰχ": 79, + "ʲ": 80, + "ɟ": 81, + "ɳ": 82, + "bʰ": 83, + "dʰ": 84 + }, + "unk_token": "UNK" + } +} \ No newline at end of file diff --git a/tokenizer_config.json b/tokenizer_config.json new file mode 100644 index 0000000..fe3ab68 --- /dev/null +++ b/tokenizer_config.json @@ -0,0 +1,37 @@ +{ + "add_prefix_space": false, + "added_tokens_decoder": { + "0": { + "content": "UNK", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "1": { + "content": "PAD", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "2": { + "content": "UTT_BOUNDARY", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + } + }, + "bos_token": "UTT_BOUNDARY", + "clean_up_tokenization_spaces": false, + "eos_token": "UTT_BOUNDARY", + "extra_special_tokens": {}, + "model_max_length": 1000000000000000019884624838656, + "pad_token": "PAD", + "tokenizer_class": "GPT2Tokenizer", + "unk_token": "UNK" +} diff --git a/train_rank0.log b/train_rank0.log new file mode 100644 index 0000000..41868e2 --- /dev/null +++ b/train_rank0.log @@ -0,0 +1,184 @@ +07/16/2026 22:47:16 - INFO - __main__ - [RANK 0] Distributed environment: DistributedType.NO +Num processes: 1 +Process index: 0 +Local process index: 0 +Device: cuda + +Mixed precision type: no + +07/16/2026 22:47:16 - INFO - __main__ - [RANK 0] Training args: +{'data_dir': 'data/phone/deu-eng/tokenized_interleaved', 'model_name_or_path': None, 'config_name': 'gpt_base_config.json', 'tokenizer_name': 'nikitastheo/phonelm-phone-deu-eng-tokenizer', 'per_device_train_batch_size': 32, 'per_device_eval_batch_size': 32, 'learning_rate': 0.0005, 'weight_decay': 0.01, 'max_steps': 40000, 'gradient_accumulation_steps': 1, 'lr_scheduler_type': , 'warmup_steps': 4000, 'output_dir': 'models/phonebabylm-phone-deu-eng-interleaved', 'seed': 42, 'model_type': None, 'block_size': None, 'push_to_hub': True, 'hub_model_id': 'nikitastheo/phonebabylm-phone-deu-eng-interleaved', 'hub_token': None, 'private': False, 'trust_remote_code': False, 'with_tracking': True, 'tracking_dir': None, 'report_to': 'wandb', 'restart_lr_at_switch': False, 'language_switch_steps': None, 'model_revision': 'main', 'config_overrides': None, 'use_slow_tokenizer': False, 'override_vocab_size': None, 'dtype': None, 'logging_steps': 10, 'cache_dir': None, 'eval_steps': 2000, 'max_grad_norm': 1.0, 'adam_epsilon': 1e-06, 'run_name': 'phonebabylm-phone-deu-eng-interleaved', 'max_train_samples': None, 'max_eval_samples': None} +07/16/2026 22:47:17 - INFO - transformers.configuration_utils - loading configuration file gpt_base_config.json +07/16/2026 22:47:17 - INFO - transformers.configuration_utils - Model config GPT2Config { + "activation_function": "gelu", + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.1, + "bos_token_id": 50256, + "embd_pdrop": 0.1, + "eos_token_id": 50256, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_ctx": 512, + "n_embd": 768, + "n_head": 12, + "n_inner": 3072, + "n_layer": 12, + "n_positions": 512, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.1, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "transformers_version": "4.57.6", + "use_cache": true, + "vocab_size": 1 +} + +07/16/2026 22:47:17 - INFO - transformers.tokenization_utils_base - loading file vocab.json from cache at /home/geofila/.cache/huggingface/hub/models--nikitastheo--phonelm-phone-deu-eng-tokenizer/snapshots/a82d9c161c2c55d913c3f8dae7f59fbdd01bd623/vocab.json +07/16/2026 22:47:17 - INFO - transformers.tokenization_utils_base - loading file merges.txt from cache at None +07/16/2026 22:47:17 - INFO - transformers.tokenization_utils_base - loading file tokenizer.json from cache at /home/geofila/.cache/huggingface/hub/models--nikitastheo--phonelm-phone-deu-eng-tokenizer/snapshots/a82d9c161c2c55d913c3f8dae7f59fbdd01bd623/tokenizer.json +07/16/2026 22:47:17 - INFO - transformers.tokenization_utils_base - loading file added_tokens.json from cache at None +07/16/2026 22:47:17 - INFO - transformers.tokenization_utils_base - loading file special_tokens_map.json from cache at /home/geofila/.cache/huggingface/hub/models--nikitastheo--phonelm-phone-deu-eng-tokenizer/snapshots/a82d9c161c2c55d913c3f8dae7f59fbdd01bd623/special_tokens_map.json +07/16/2026 22:47:17 - INFO - transformers.tokenization_utils_base - loading file tokenizer_config.json from cache at /home/geofila/.cache/huggingface/hub/models--nikitastheo--phonelm-phone-deu-eng-tokenizer/snapshots/a82d9c161c2c55d913c3f8dae7f59fbdd01bd623/tokenizer_config.json +07/16/2026 22:47:17 - INFO - transformers.tokenization_utils_base - loading file chat_template.jinja from cache at None +07/16/2026 22:47:19 - INFO - transformers.generation.configuration_utils - Generate config GenerationConfig { + "bos_token_id": 2, + "eos_token_id": 2, + "pad_token_id": 1 +} + +07/16/2026 22:47:20 - INFO - __main__ - [RANK 0] Model parameters: 85.51M total (85.51M trainable) +07/16/2026 22:47:39 - INFO - __main__ - [RANK 0] Sample 167621 of the training set: {'input_ids': tensor([ 5, 11, 30, 24, 52, 5, 56, 15, 5, 26, 30, 13, 25, 3, 31, 5, 35, 54, + 22, 15, 25, 14, 5, 3, 16, 56, 15, 5, 45, 24, 30, 14, 5, 56, 15, 5, + 23, 52, 46, 18, 5, 22, 28, 8, 21, 14, 5, 54, 6, 56, 15, 5, 26, 28, + 8, 30, 3, 31, 5, 35, 54, 6, 15, 52, 25, 5, 45, 3, 56, 5, 19, 17, + 35, 26, 5, 11, 52, 54, 21, 5, 45, 3, 48, 5, 45, 71, 8, 5, 26, 15, + 26, 38, 32, 16, 25, 3, 25, 5, 11, 30, 3, 48, 3, 14, 5, 54, 6, 5, + 22, 46, 35, 15, 16, 5, 22, 54, 16, 48, 3, 14, 5, 54, 6, 5, 25, 52, + 13, 25, 5, 71, 8, 22, 14, 5, 26, 18, 52, 3, 31, 14, 5, 54, 6, 5, + 54, 16, 59, 16, 14, 5, 70, 16, 25, 5, 28, 8, 6, 15, 16, 5, 21, 3, + 26, 18, 15, 52, 5, 22, 3, 31, 35, 26, 3, 14, 5, 11, 3, 27, 3, 31, + 22, 34, 8, 18, 26, 5, 24, 30, 5, 45, 71, 8, 5, 16, 34, 8, 5, 18, + 15, 5, 19, 71, 8, 52, 5, 3, 16, 18, 52, 32, 26, 18, 3, 25, 5, 23, + 46, 14, 5, 70, 16, 25, 5, 19, 71, 8, 5, 35, 45, 3, 35, 5, 13, 14, + 5, 18, 17, 35, 5, 3, 16, 5, 32, 6, 52, 3, 60, 3, 31, 5, 11, 52, + 54, 21, 56, 15, 5, 23, 54, 16, 52, 70, 35, 5, 73, 6, 15, 52, 5, 56, + 15, 5, 25, 52, 32, 26, 15, 52, 5, 18, 15, 5, 56, 15, 5, 48, 13, 16, + 15, 5, 25, 72, 23, 14, 5, 24, 16, 56, 15, 5, 48, 3, 21, 16, 28, 38, + 28, 8, 26, 5, 56, 15, 5, 35, 3, 48, 15, 16, 5, 45, 54, 14, 5, 26, + 73, 5, 30, 72, 52, 49, 5, 56, 70, 18, 5, 19, 70, 11, 5, 54, 6, 5, + 3, 18, 5, 26, 28, 8, 21, 25, 5, 18, 15, 22, 28, 5, 52, 3, 14, 71, + 8, 6, 25, 5, 70, 14, 5, 54, 5, 38, 72, 52, 30, 15, 52, 5, 56, 32, + 52, 45, 54, 14, 5, 54, 5, 26, 35, 45, 32, 52, 5, 54, 6, 5, 35, 72, + 52, 38, 3, 18, 5, 30, 46, 25, 5, 25, 10, 16, 5, 70, 18, 5, 45, 54, + 16, 5, 32, 16, 25, 5, 15, 38, 72, 16, 5, 45, 3, 48, 5, 26, 18, 17, + 25, 5, 54, 5, 52, 10, 16, 25, 5, 18, 46, 22, 15, 30, 5, 26, 38, 52, + 32, 25, 5, 45, 3, 56, 5, 21, 3, 26, 3, 14, 5, 22, 3, 31, 35, 26, + 3, 14, 5, 6, 32, 52, 28, 5, 22, 32, 26, 18, 5, 48, 13, 16, 15, 5, + 18, 28, 8, 26, 71, 8, 6, 3, 26, 5, 70, 16, 25, 5, 54, 5, 26, 15, + 38, 30, 13, 5, 54, 6, 5, 25])} WORD_BOUNDARY f l ɔ ɹ WORD_BOUNDARY ð ə WORD_BOUNDARY s l aɪ d ɪ ŋ WORD_BOUNDARY k ʌ b ə d z WORD_BOUNDARY ɪ n ð ə WORD_BOUNDARY w ɔ l z WORD_BOUNDARY ð ə WORD_BOUNDARY ɡ ɹ eɪ t WORD_BOUNDARY b i ː m z WORD_BOUNDARY ʌ v ð ə WORD_BOUNDARY s i ː l ɪ ŋ WORD_BOUNDARY k ʌ v ə ɹ d WORD_BOUNDARY w ɪ ð WORD_BOUNDARY h ʊ k s WORD_BOUNDARY f ɹ ʌ m WORD_BOUNDARY w ɪ t̠ʃ WORD_BOUNDARY w ɜ ː WORD_BOUNDARY s ə s p ɛ n d ɪ d WORD_BOUNDARY f l ɪ t̠ʃ ɪ z WORD_BOUNDARY ʌ v WORD_BOUNDARY b eɪ k ə n WORD_BOUNDARY b ʌ n t̠ʃ ɪ z WORD_BOUNDARY ʌ v WORD_BOUNDARY d ɹ aɪ d WORD_BOUNDARY ɜ ː b z WORD_BOUNDARY s t ɹ ɪ ŋ z WORD_BOUNDARY ʌ v WORD_BOUNDARY ʌ n iə n z WORD_BOUNDARY æ n d WORD_BOUNDARY i ː v ə n WORD_BOUNDARY m ɪ s t ə ɹ WORD_BOUNDARY b ɪ ŋ k s ɪ z WORD_BOUNDARY f ɪ ʃ ɪ ŋ b u ː t s WORD_BOUNDARY ɔ l WORD_BOUNDARY w ɜ ː WORD_BOUNDARY n u ː WORD_BOUNDARY t ə WORD_BOUNDARY h ɜ ː ɹ WORD_BOUNDARY ɪ n t ɹ ɛ s t ɪ d WORD_BOUNDARY ɡ eɪ z WORD_BOUNDARY æ n d WORD_BOUNDARY h ɜ ː WORD_BOUNDARY k w ɪ k WORD_BOUNDARY aɪ z WORD_BOUNDARY t ʊ k WORD_BOUNDARY ɪ n WORD_BOUNDARY ɛ v ɹ ɪ θ ɪ ŋ WORD_BOUNDARY f ɹ ʌ m ð ə WORD_BOUNDARY ɡ ʌ n ɹ æ k WORD_BOUNDARY oʊ v ə ɹ WORD_BOUNDARY ð ə WORD_BOUNDARY d ɹ ɛ s ə ɹ WORD_BOUNDARY t ə WORD_BOUNDARY ð ə WORD_BOUNDARY t̠ʃ aɪ n ə WORD_BOUNDARY d ɑ ɡ z WORD_BOUNDARY ɔ n ð ə WORD_BOUNDARY t̠ʃ ɪ m n i p i ː s WORD_BOUNDARY ð ə WORD_BOUNDARY k ɪ t̠ʃ ə n WORD_BOUNDARY w ʌ z WORD_BOUNDARY s oʊ WORD_BOUNDARY l ɑ ɹ d̠ʒ WORD_BOUNDARY ð æ t WORD_BOUNDARY h æ f WORD_BOUNDARY ʌ v WORD_BOUNDARY ɪ t WORD_BOUNDARY s i ː m d WORD_BOUNDARY t ə b i WORD_BOUNDARY ɹ ɪ z ɜ ː v d WORD_BOUNDARY æ z WORD_BOUNDARY ʌ WORD_BOUNDARY p ɑ ɹ l ə ɹ WORD_BOUNDARY ð ɛ ɹ w ʌ z WORD_BOUNDARY ʌ WORD_BOUNDARY s k w ɛ ɹ WORD_BOUNDARY ʌ v WORD_BOUNDARY k ɑ ɹ p ɪ t WORD_BOUNDARY l eɪ d WORD_BOUNDARY d aʊ n WORD_BOUNDARY æ t WORD_BOUNDARY w ʌ n WORD_BOUNDARY ɛ n d WORD_BOUNDARY ə p ɑ n WORD_BOUNDARY w ɪ t̠ʃ WORD_BOUNDARY s t ʊ d WORD_BOUNDARY ʌ WORD_BOUNDARY ɹ aʊ n d WORD_BOUNDARY t eɪ b ə l WORD_BOUNDARY s p ɹ ɛ d WORD_BOUNDARY w ɪ ð WORD_BOUNDARY m ɪ s ɪ z WORD_BOUNDARY b ɪ ŋ k s ɪ z WORD_BOUNDARY v ɛ ɹ i WORD_BOUNDARY b ɛ s t WORD_BOUNDARY t̠ʃ aɪ n ə WORD_BOUNDARY t i ː s ɜ ː v ɪ s WORD_BOUNDARY æ n d WORD_BOUNDARY ʌ WORD_BOUNDARY s ə p l aɪ WORD_BOUNDARY ʌ v WORD_BOUNDARY d +07/16/2026 22:47:39 - INFO - __main__ - [RANK 0] Sample 29184 of the training set: {'input_ids': tensor([13, 16, 15, 5, 18, 7, 27, 15, 16, 5, 21, 3, 18, 5, 35, 17, 38, 11, + 9, 23, 32, 30, 18, 5, 11, 32, 9, 27, 30, 40, 8, 26, 5, 25, 28, 8, + 5, 35, 3, 26, 18, 15, 5, 14, 32, 20, 18, 15, 5, 25, 33, 8, 16, 5, + 19, 17, 16, 18, 5, 6, 28, 8, 25, 9, 5, 19, 3, 16, 10, 11, 5, 17, + 16, 18, 5, 23, 3, 31, 5, 3, 16, 5, 25, 7, 26, 5, 7, 16, 25, 15, + 12, 15, 5, 20, 3, 21, 9, 5, 38, 24, 20, 18, 10, 14, 15, 16, 18, 5, + 25, 7, 8, 5, 14, 7, 8, 26, 5, 25, 32, 9, 5, 19, 17, 16, 18, 5, + 21, 3, 18, 5, 10, 23, 15, 16, 5, 14, 40, 8, 5, 23, 9, 40, 8, 26, + 5, 6, 28, 8, 5, 21, 41, 8, 30, 12, 32, 8, 25, 9, 5, 2, 23, 30, + 24, 20, 5, 21, 3, 4, 5, 16, 3, 4, 18, 5, 14, 40, 8, 5, 7, 16, + 5, 14, 7, 8, 23, 18, 15, 5, 25, 32, 9, 5, 14, 24, 30, 25, 7, 8, + 18, 5, 17, 16, 18, 5, 14, 32, 20, 18, 15, 5, 25, 33, 8, 16, 5, 19, + 17, 16, 18, 5, 10, 11, 5, 25, 28, 8, 5, 27, 36, 9, 20, 15, 5, 7, + 30, 26, 5, 32, 9, 5, 7, 8, 22, 9, 5, 25, 7, 26, 5, 11, 28, 8, + 30, 15, 5, 14, 3, 30, 22, 9, 23, 32, 30, 18, 5, 14, 7, 8, 5, 6, + 7, 9, 11, 5, 32, 9, 5, 7, 30, 15, 26, 5, 35, 17, 38, 11, 9, 23, + 32, 30, 18, 5, 11, 24, 9, 18, 5, 17, 16, 18, 5, 11, 36, 30, 18, 15, + 5, 14, 3, 4, 5, 25, 28, 8, 5, 18, 7, 27, 15, 16, 5, 17, 16, 18, + 5, 25, 33, 8, 16, 5, 18, 24, 9, 16, 3, 26, 18, 9, 5, 21, 3, 18, + 5, 14, 3, 30, 22, 9, 5, 25, 7, 16, 5, 23, 3, 31, 5, 32, 9, 5, + 3, 16, 5, 25, 28, 8, 5, 25, 9, 3, 18, 15, 5, 35, 7, 21, 9, 5, + 6, 40, 8, 5, 25, 32, 9, 5, 19, 17, 16, 18, 5, 6, 7, 8, 9, 5, + 21, 3, 18, 5, 10, 23, 15, 16, 5, 14, 40, 8, 5, 23, 9, 40, 8, 26, + 5, 6, 28, 8, 5, 13, 16, 5, 12, 17, 16, 25, 9, 5, 18, 29, 21, 5, + 2, 23, 34, 8, 18, 15, 16, 5, 7, 8, 22, 15, 16, 18, 5, 14, 7, 8, + 23, 18, 15, 5, 25, 32, 9, 5, 14, 24, 30, 25, 7, 8, 18, 5, 19, 40, + 8, 38, 5, 25, 33, 8, 16, 5, 19, 17, 16, 18, 5, 19, 32, 12, 17, 16, + 18, 9, 5, 17, 16, 18, 5, 42, 11, 16, 15, 18, 15, 5, 25, 28, 8, 5, + 35, 3, 26, 18, 15, 5, 6, 7])} aɪ n ə WORD_BOUNDARY t a ʃ ə n WORD_BOUNDARY m ɪ t WORD_BOUNDARY k ʊ p f ɐ ɡ ɛ l t WORD_BOUNDARY f ɛ ɐ ʃ l o ː s WORD_BOUNDARY d i ː WORD_BOUNDARY k ɪ s t ə WORD_BOUNDARY z ɛ ts t ə WORD_BOUNDARY d e ː n WORD_BOUNDARY h ʊ n t WORD_BOUNDARY v i ː d ɐ WORD_BOUNDARY h ɪ n aʊ f WORD_BOUNDARY ʊ n t WORD_BOUNDARY ɡ ɪ ŋ WORD_BOUNDARY ɪ n WORD_BOUNDARY d a s WORD_BOUNDARY a n d ə ʀ ə WORD_BOUNDARY ts ɪ m ɐ WORD_BOUNDARY p ɔ ts t aʊ z ə n t WORD_BOUNDARY d a ː WORD_BOUNDARY z a ː s WORD_BOUNDARY d ɛ ɐ WORD_BOUNDARY h ʊ n t WORD_BOUNDARY m ɪ t WORD_BOUNDARY aʊ ɡ ə n WORD_BOUNDARY z o ː WORD_BOUNDARY ɡ ɐ o ː s WORD_BOUNDARY v i ː WORD_BOUNDARY m y ː l ʀ ɛ ː d ɐ WORD_BOUNDARY UTT_BOUNDARY ɡ l ɔ ts WORD_BOUNDARY m ɪ ç WORD_BOUNDARY n ɪ ç t WORD_BOUNDARY z o ː WORD_BOUNDARY a n WORD_BOUNDARY z a ː ɡ t ə WORD_BOUNDARY d ɛ ɐ WORD_BOUNDARY z ɔ l d a ː t WORD_BOUNDARY ʊ n t WORD_BOUNDARY z ɛ ts t ə WORD_BOUNDARY d e ː n WORD_BOUNDARY h ʊ n t WORD_BOUNDARY aʊ f WORD_BOUNDARY d i ː WORD_BOUNDARY ʃ ʏ ɐ ts ə WORD_BOUNDARY a l s WORD_BOUNDARY ɛ ɐ WORD_BOUNDARY a ː b ɐ WORD_BOUNDARY d a s WORD_BOUNDARY f i ː l ə WORD_BOUNDARY z ɪ l b ɐ ɡ ɛ l t WORD_BOUNDARY z a ː WORD_BOUNDARY v a ɐ f WORD_BOUNDARY ɛ ɐ WORD_BOUNDARY a l ə s WORD_BOUNDARY k ʊ p f ɐ ɡ ɛ l t WORD_BOUNDARY f ɔ ɐ t WORD_BOUNDARY ʊ n t WORD_BOUNDARY f ʏ l t ə WORD_BOUNDARY z ɪ ç WORD_BOUNDARY d i ː WORD_BOUNDARY t a ʃ ə n WORD_BOUNDARY ʊ n t WORD_BOUNDARY d e ː n WORD_BOUNDARY t ɔ ɐ n ɪ s t ɐ WORD_BOUNDARY m ɪ t WORD_BOUNDARY z ɪ l b ɐ WORD_BOUNDARY d a n WORD_BOUNDARY ɡ ɪ ŋ WORD_BOUNDARY ɛ ɐ WORD_BOUNDARY ɪ n WORD_BOUNDARY d i ː WORD_BOUNDARY d ɐ ɪ t ə WORD_BOUNDARY k a m ɐ WORD_BOUNDARY v o ː WORD_BOUNDARY d ɛ ɐ WORD_BOUNDARY h ʊ n t WORD_BOUNDARY v a ː ɐ WORD_BOUNDARY m ɪ t WORD_BOUNDARY aʊ ɡ ə n WORD_BOUNDARY z o ː WORD_BOUNDARY ɡ ɐ o ː s WORD_BOUNDARY v i ː WORD_BOUNDARY aɪ n WORD_BOUNDARY ʀ ʊ n d ɐ WORD_BOUNDARY t ʊɐ m WORD_BOUNDARY UTT_BOUNDARY ɡ u ː t ə n WORD_BOUNDARY a ː b ə n t WORD_BOUNDARY z a ː ɡ t ə WORD_BOUNDARY d ɛ ɐ WORD_BOUNDARY z ɔ l d a ː t WORD_BOUNDARY h o ː p WORD_BOUNDARY d e ː n WORD_BOUNDARY h ʊ n t WORD_BOUNDARY h ɛ ʀ ʊ n t ɐ WORD_BOUNDARY ʊ n t WORD_BOUNDARY œ f n ə t ə WORD_BOUNDARY d i ː WORD_BOUNDARY k ɪ s t ə WORD_BOUNDARY v a +07/16/2026 22:47:39 - INFO - __main__ - [RANK 0] Sample 6556 of the training set: {'input_ids': tensor([16, 5, 2, 6, 32, 8, 12, 15, 16, 18, 5, 25, 32, 9, 5, 27, 34, 8, + 30, 11, 33, 8, 12, 28, 8, 15, 16, 5, 11, 28, 8, 30, 5, 10, 39, 5, + 25, 28, 8, 5, 14, 24, 16, 18, 7, 35, 26, 4, 34, 8, 30, 15, 5, 10, + 26, 5, 18, 9, 24, 20, 25, 33, 8, 21, 5, 6, 7, 8, 9, 5, 19, 24, + 36, 18, 5, 7, 30, 15, 26, 5, 11, 12, 41, 8, 20, 13, 18, 3, 4, 5, + 3, 16, 5, 25, 32, 9, 5, 35, 3, 9, 4, 15, 5, 25, 7, 26, 5, 10, + 11, 12, 33, 8, 23, 15, 16, 25, 15, 5, 32, 9, 13, 23, 16, 3, 26, 5, + 6, 29, 25, 15, 5, 30, 33, 8, 38, 19, 7, 11, 18, 5, 32, 9, 42, 9, + 18, 9, 18, 5, 32, 26, 5, 6, 29, 25, 15, 5, 32, 9, 20, 32, 8, 30, + 18, 5, 25, 7, 26, 5, 16, 24, 39, 5, 35, 13, 16, 15, 5, 27, 38, 34, + 8, 9, 5, 11, 24, 16, 5, 25, 33, 8, 16, 5, 30, 7, 16, 18, 27, 18, + 9, 13, 4, 9, 16, 5, 23, 15, 11, 17, 16, 25, 15, 16, 5, 6, 24, 9, + 25, 15, 16, 5, 14, 13, 5, 7, 30, 26, 5, 25, 28, 8, 5, 38, 9, 33, + 8, 25, 3, 4, 18, 5, 20, 34, 8, 5, 32, 16, 25, 15, 5, 6, 7, 8, + 9, 5, 23, 3, 31, 5, 25, 28, 8, 5, 11, 12, 10, 5, 25, 32, 26, 5, + 12, 3, 4, 18, 9, 26, 5, 18, 7, 48, 9, 5, 10, 11, 5, 11, 12, 10, + 5, 19, 7, 9, 38, 9, 5, 20, 34, 8, 5, 25, 28, 8, 5, 21, 3, 18, + 5, 25, 32, 9, 5, 23, 9, 40, 8, 26, 15, 16, 5, 21, 32, 31, 15, 5, + 25, 33, 8, 16, 5, 23, 7, 31, 5, 19, 3, 16, 17, 16, 18, 9, 27, 12, + 3, 18, 5, 17, 16, 18, 5, 14, 7, 8, 23, 18, 15, 5, 6, 3, 30, 5, + 21, 13, 16, 15, 5, 22, 32, 35, 28, 8, 5, 25, 32, 16, 5, 25, 33, 8, + 16, 5, 23, 7, 16, 20, 15, 16, 5, 18, 7, 8, 35, 5, 27, 30, 7, 8, + 11, 15, 16, 5, 19, 7, 8, 38, 26, 5, 21, 28, 8, 9, 5, 7, 8, 22, + 9, 5, 6, 40, 8, 30, 5, 23, 15, 25, 7, 39, 18, 5, 25, 7, 26, 5, + 14, 28, 8, 5, 18, 24, 25, 21, 41, 8, 25, 15, 5, 14, 13, 16, 5, 6, + 36, 9, 25, 15, 5, 2, 28, 8, 12, 15, 5, 22, 32, 35, 28, 8, 5, 2, + 11, 12, 13, 30, 3, 4, 5, 21, 3, 18, 5, 32, 9, 27, 12, 24, 35, 15, + 16, 15, 21, 5, 22, 30, 3, 35, 5, 22, 30, 28, 8, 38, 5, 14, 28, 8, + 5, 25, 32, 16, 5, 7, 8, 22])} n WORD_BOUNDARY UTT_BOUNDARY v ɛ ː ʀ ə n t WORD_BOUNDARY d ɛ ɐ WORD_BOUNDARY ʃ u ː l f e ː ʀ i ː ə n WORD_BOUNDARY f i ː l WORD_BOUNDARY aʊ x WORD_BOUNDARY d i ː WORD_BOUNDARY z ɔ n t a k s ç u ː l ə WORD_BOUNDARY aʊ s WORD_BOUNDARY t ɐ ɔ ts d e ː m WORD_BOUNDARY v a ː ɐ WORD_BOUNDARY h ɔ ʏ t WORD_BOUNDARY a l ə s WORD_BOUNDARY f ʀ y ː ts aɪ t ɪ ç WORD_BOUNDARY ɪ n WORD_BOUNDARY d ɛ ɐ WORD_BOUNDARY k ɪ ɐ ç ə WORD_BOUNDARY d a s WORD_BOUNDARY aʊ f ʀ e ː ɡ ə n d ə WORD_BOUNDARY ɛ ɐ aɪ ɡ n ɪ s WORD_BOUNDARY v ʊɐ d ə WORD_BOUNDARY l e ː p h a f t WORD_BOUNDARY ɛ ɐ œ ɐ t ɐ t WORD_BOUNDARY ɛ s WORD_BOUNDARY v ʊɐ d ə WORD_BOUNDARY ɛ ɐ ts ɛ ː l t WORD_BOUNDARY d a s WORD_BOUNDARY n ɔ x WORD_BOUNDARY k aɪ n ə WORD_BOUNDARY ʃ p u ː ɐ WORD_BOUNDARY f ɔ n WORD_BOUNDARY d e ː n WORD_BOUNDARY l a n t ʃ t ɐ aɪ ç ɐ n WORD_BOUNDARY ɡ ə f ʊ n d ə n WORD_BOUNDARY v ɔ ɐ d ə n WORD_BOUNDARY z aɪ WORD_BOUNDARY a l s WORD_BOUNDARY d i ː WORD_BOUNDARY p ɐ e ː d ɪ ç t WORD_BOUNDARY ts u ː WORD_BOUNDARY ɛ n d ə WORD_BOUNDARY v a ː ɐ WORD_BOUNDARY ɡ ɪ ŋ WORD_BOUNDARY d i ː WORD_BOUNDARY f ʀ aʊ WORD_BOUNDARY d ɛ s WORD_BOUNDARY ʀ ɪ ç t ɐ s WORD_BOUNDARY t a t̠ʃ ɐ WORD_BOUNDARY aʊ f WORD_BOUNDARY f ʀ aʊ WORD_BOUNDARY h a ɐ p ɐ WORD_BOUNDARY ts u ː WORD_BOUNDARY d i ː WORD_BOUNDARY m ɪ t WORD_BOUNDARY d ɛ ɐ WORD_BOUNDARY ɡ ɐ o ː s ə n WORD_BOUNDARY m ɛ ŋ ə WORD_BOUNDARY d e ː n WORD_BOUNDARY ɡ a ŋ WORD_BOUNDARY h ɪ n ʊ n t ɐ ʃ ʀ ɪ t WORD_BOUNDARY ʊ n t WORD_BOUNDARY z a ː ɡ t ə WORD_BOUNDARY v ɪ l WORD_BOUNDARY m aɪ n ə WORD_BOUNDARY b ɛ k i ː WORD_BOUNDARY d ɛ n WORD_BOUNDARY d e ː n WORD_BOUNDARY ɡ a n ts ə n WORD_BOUNDARY t a ː k WORD_BOUNDARY ʃ l a ː f ə n WORD_BOUNDARY h a ː p s WORD_BOUNDARY m i ː ɐ WORD_BOUNDARY a ː b ɐ WORD_BOUNDARY v o ː l WORD_BOUNDARY ɡ ə d a x t WORD_BOUNDARY d a s WORD_BOUNDARY z i ː WORD_BOUNDARY t ɔ d m y ː d ə WORD_BOUNDARY z aɪ n WORD_BOUNDARY v ʏ ɐ d ə WORD_BOUNDARY UTT_BOUNDARY i ː ʀ ə WORD_BOUNDARY b ɛ k i ː WORD_BOUNDARY UTT_BOUNDARY f ʀ aɪ l ɪ ç WORD_BOUNDARY m ɪ t WORD_BOUNDARY ɛ ɐ ʃ ʀ ɔ k ə n ə m WORD_BOUNDARY b l ɪ k WORD_BOUNDARY b l i ː p WORD_BOUNDARY z i ː WORD_BOUNDARY d ɛ n WORD_BOUNDARY a ː b +07/16/2026 22:47:42 - INFO - __main__ - [RANK 0] ***** Running training ***** +07/16/2026 22:47:42 - INFO - __main__ - [RANK 0] Num examples = 203590 +07/16/2026 22:47:42 - INFO - __main__ - [RANK 0] Num Training Steps = 40000 +07/16/2026 22:47:42 - INFO - __main__ - [RANK 0] Instantaneous batch size per device = 32 +07/16/2026 22:47:42 - INFO - __main__ - [RANK 0] Total train batch size (w. parallel, distributed & accumulation) = 32 +07/16/2026 22:47:42 - INFO - __main__ - [RANK 0] Gradient Accumulation steps = 1 +07/16/2026 22:47:42 - INFO - __main__ - [RANK 0] Total optimization steps = 40000 +07/16/2026 22:47:42 - INFO - transformers.configuration_utils - Configuration saved in models/phonebabylm-phone-deu-eng-interleaved/config.json +07/16/2026 22:47:42 - INFO - transformers.generation.configuration_utils - Configuration saved in models/phonebabylm-phone-deu-eng-interleaved/generation_config.json +07/16/2026 22:47:45 - INFO - transformers.modeling_utils - Model weights saved in models/phonebabylm-phone-deu-eng-interleaved/model.safetensors +07/16/2026 22:47:45 - INFO - transformers.tokenization_utils_base - tokenizer config file saved in models/phonebabylm-phone-deu-eng-interleaved/tokenizer_config.json +07/16/2026 22:47:45 - INFO - transformers.tokenization_utils_base - Special tokens file saved in models/phonebabylm-phone-deu-eng-interleaved/special_tokens_map.json +07/16/2026 22:47:50 - INFO - __main__ - [RANK 0] Starting epoch 0... +07/16/2026 22:47:51 - WARNING - transformers.modeling_utils - `loss_type=None` was set in the config but it is unrecognized. Using the default loss: `ForCausalLMLoss`. +07/16/2026 22:58:50 - INFO - __main__ - [RANK 0] perplexity: 3.4981748920090556 eval_loss: 1.2522413730621338 accuracy: 0.6286 +07/16/2026 23:09:55 - INFO - __main__ - [RANK 0] perplexity: 3.090224323166524 eval_loss: 1.1282436847686768 accuracy: 0.6635 +07/16/2026 23:20:59 - INFO - __main__ - [RANK 0] perplexity: 2.8887235844293055 eval_loss: 1.0608147382736206 accuracy: 0.6829 +07/16/2026 23:22:58 - INFO - __main__ - [RANK 0] Starting epoch 1... +07/16/2026 23:32:04 - INFO - __main__ - [RANK 0] perplexity: 2.7866070328443007 eval_loss: 1.0248247385025024 accuracy: 0.6932 +07/16/2026 23:43:09 - INFO - __main__ - [RANK 0] perplexity: 2.717227754278576 eval_loss: 0.9996121525764465 accuracy: 0.7002 +07/16/2026 23:54:14 - INFO - __main__ - [RANK 0] perplexity: 2.662045793634297 eval_loss: 0.979094922542572 accuracy: 0.7061 +07/16/2026 23:58:13 - INFO - __main__ - [RANK 0] Starting epoch 2... +07/17/2026 00:05:19 - INFO - __main__ - [RANK 0] perplexity: 2.6255481989944776 eval_loss: 0.9652897119522095 accuracy: 0.7102 +07/17/2026 00:16:23 - INFO - __main__ - [RANK 0] perplexity: 2.5913564344248097 eval_loss: 0.9521814584732056 accuracy: 0.7138 +07/17/2026 00:27:28 - INFO - __main__ - [RANK 0] perplexity: 2.5649778532126026 eval_loss: 0.9419498443603516 accuracy: 0.7169 +07/17/2026 00:33:26 - INFO - __main__ - [RANK 0] Starting epoch 3... +07/17/2026 00:38:43 - INFO - __main__ - [RANK 0] perplexity: 2.5450176439772125 eval_loss: 0.9341375827789307 accuracy: 0.7196 +07/17/2026 00:50:44 - INFO - __main__ - [RANK 0] perplexity: 2.523804941363755 eval_loss: 0.9257676601409912 accuracy: 0.7216 +07/17/2026 01:02:02 - INFO - __main__ - [RANK 0] perplexity: 2.504542042072227 eval_loss: 0.9181059002876282 accuracy: 0.7241 +07/17/2026 01:10:00 - INFO - __main__ - [RANK 0] Starting epoch 4... +07/17/2026 01:13:06 - INFO - __main__ - [RANK 0] perplexity: 2.4924423233459936 eval_loss: 0.9132630825042725 accuracy: 0.7257 +07/17/2026 01:24:11 - INFO - __main__ - [RANK 0] perplexity: 2.4772802757896057 eval_loss: 0.907161295413971 accuracy: 0.7277 +07/17/2026 01:35:16 - INFO - __main__ - [RANK 0] perplexity: 2.4638050764913904 eval_loss: 0.9017069339752197 accuracy: 0.7294 +07/17/2026 01:45:13 - INFO - __main__ - [RANK 0] Starting epoch 5... +07/17/2026 01:46:20 - INFO - __main__ - [RANK 0] perplexity: 2.4591889317422853 eval_loss: 0.8998315930366516 accuracy: 0.7302 +07/17/2026 01:57:25 - INFO - __main__ - [RANK 0] perplexity: 2.44883489662295 eval_loss: 0.895612359046936 accuracy: 0.7316 +07/17/2026 02:08:31 - INFO - __main__ - [RANK 0] perplexity: 2.4370661854301496 eval_loss: 0.8907949328422546 accuracy: 0.7328 +07/17/2026 02:19:36 - INFO - __main__ - [RANK 0] perplexity: 2.4286724169445133 eval_loss: 0.8873447775840759 accuracy: 0.7338 +07/17/2026 02:20:35 - INFO - __main__ - [RANK 0] Starting epoch 6... +07/17/2026 02:30:41 - INFO - __main__ - [RANK 0] perplexity: 2.432660355583435 eval_loss: 0.8889854550361633 accuracy: 0.7341 diff --git a/training_metadata.json b/training_metadata.json new file mode 100644 index 0000000..d3c3f7e --- /dev/null +++ b/training_metadata.json @@ -0,0 +1,25 @@ +{ + "input_config": { + "model_type": "gpt2", + "architectures": [ + "GPT2LMHeadModel" + ], + "activation_function": "gelu", + "attn_pdrop": 0.1, + "embd_pdrop": 0.1, + "resid_pdrop": 0.1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "n_embd": 768, + "n_inner": 3072, + "n_head": 12, + "n_layer": 12, + "vocab_size": 1, + "n_ctx": 512, + "n_positions": 512 + }, + "num_steps_switch": null, + "save_steps": [ + 40000 + ] +} \ No newline at end of file diff --git a/vocab.json b/vocab.json new file mode 100644 index 0000000..d67603b --- /dev/null +++ b/vocab.json @@ -0,0 +1 @@ +{"UNK":0,"PAD":1,"UTT_BOUNDARY":2,"ɪ":3,"ç":4,"WORD_BOUNDARY":5,"v":6,"a":7,"ː":8,"ɐ":9,"aʊ":10,"f":11,"ʀ":12,"aɪ":13,"z":14,"ə":15,"n":16,"ʊ":17,"t":18,"h":19,"ts":20,"m":21,"b":22,"ɡ":23,"ɔ":24,"d":25,"s":26,"ʃ":27,"i":28,"ʊɐ":29,"l":30,"ŋ":31,"ɛ":32,"e":33,"u":34,"k":35,"ʏ":36,"j":37,"p":38,"x":39,"o":40,"y":41,"œ":42,"pf":43,"ø":44,"w":45,"eɪ":46,"ɒ":47,"t̠ʃ":48,"d̠ʒ":49,"ã":50,"əʊ":51,"ɹ":52,"eə":53,"ʌ":54,"ɛɪ":55,"ð":56,"ɔɪ":57,"ʒ":58,"iə":59,"θ":60,"ʊə":61,"œ̃":62,"ɔ̃":63,"l̩":64,"tɕ":65,"ɕ":66,"c":67,"ɲ":68,"tʰ":69,"æ":70,"ɜ":71,"ɑ":72,"oʊ":73,"ʂ":74,"ʈ":75,"ɖ":76,"ɭ":77,"ʋ":78,"ʰχ":79,"ʲ":80,"ɟ":81,"ɳ":82,"bʰ":83,"dʰ":84} \ No newline at end of file