From d9974712a99df755b0c71e3f1f6174ed0fc4bae9 Mon Sep 17 00:00:00 2001 From: ModelHub XC Date: Sun, 30 Aug 2026 05:52:16 +0800 Subject: [PATCH] =?UTF-8?q?=E5=88=9D=E5=A7=8B=E5=8C=96=E9=A1=B9=E7=9B=AE?= =?UTF-8?q?=EF=BC=8C=E7=94=B1ModelHub=20XC=E7=A4=BE=E5=8C=BA=E6=8F=90?= =?UTF-8?q?=E4=BE=9B=E6=A8=A1=E5=9E=8B?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Model: philipperen55/Qwen3-14B-datasetCPT70-axolotl Source: Original Platform --- .gitattributes | 36 ++++++++ README.md | 190 +++++++++++++++++++++++++++++++++++++++++ axolotl_config.yml | 66 ++++++++++++++ config.json | 75 ++++++++++++++++ generation_config.json | 9 ++ model.safetensors | 3 + tokenizer.json | 3 + tokenizer_config.json | 31 +++++++ training_args.bin | 3 + 9 files changed, 416 insertions(+) create mode 100644 .gitattributes create mode 100644 README.md create mode 100644 axolotl_config.yml create mode 100644 config.json create mode 100644 generation_config.json create mode 100644 model.safetensors create mode 100644 tokenizer.json create mode 100644 tokenizer_config.json create mode 100644 training_args.bin diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..52373fe --- /dev/null +++ b/.gitattributes @@ -0,0 +1,36 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text +tokenizer.json filter=lfs diff=lfs merge=lfs -text diff --git a/README.md b/README.md new file mode 100644 index 0000000..0a9c853 --- /dev/null +++ b/README.md @@ -0,0 +1,190 @@ +--- +library_name: transformers +license: apache-2.0 +base_model: Qwen/Qwen3-14B-Base +tags: +- generated_from_trainer +datasets: +- philipperen55/datasetCPT70axolotlRandomized +model-index: +- name: outputs_cpt + results: [] +--- + + + +[Built with Axolotl](https://github.com/axolotl-ai-cloud/axolotl) +
See axolotl config + +axolotl version: `0.17.0.dev0` +```yaml +base_model: Qwen/Qwen3-14B-Base + +trust_remote_code: true +tokenizer_use_fast: true + +load_in_8bit: false +load_in_4bit: false + + +datasets: + - path: philipperen55/datasetCPT70axolotlRandomized + split: train + data_files: datasetCPT70axolotlRandomized.jsonl + type: completion + field: text + + +val_set_size: 0.0001 + +dataset_prepared_path: prepared_cpt +output_dir: outputs_cpt + +sequence_len: 2048 +pad_to_sequence_len: true +sample_packing: true +eval_sample_packing: true +attn_implementation: flash_attention_2 +excess_length_strategy: truncate +train_on_inputs: true +add_eos_token: true + +# FULL finetune CPT +micro_batch_size: 2 +gradient_accumulation_steps: 8 +num_epochs: 1 + + +optimizer: adamw_8bit +learning_rate: 4e-5 +weight_decay: 0.01 +lr_scheduler: constant_with_warmup +warmup_ratio: 0.01 +max_grad_norm: 1.0 + +fp16: false +bf16: true +tf32: true +gradient_checkpointing: false + +logging_steps: 50 +eval_steps: 1000 +save_steps: 5000 +save_total_limit: 1 +save_only_model: true + +seed: 42 +#mettre 24 si ya plus de 24 vspu, sinon mettre 16 si ya 24vcpu +dataset_num_proc: 24 + +# WandB +wandb_project: qwen3_14b_cpt_full_axolotl + +# Hub +hub_model_id: +push_to_hub: false #mon script upload de façon plus sûre +hub_strategy: every_save + +``` + +

+ +# outputs_cpt + +This model is a fine-tuned version of [Qwen/Qwen3-14B-Base](https://huggingface.co/Qwen/Qwen3-14B-Base) on the philipperen55/datasetCPT70axolotlRandomized dataset. +It achieves the following results on the evaluation set: +- Loss: 1.8920 +- Ppl: 6.6325 +- Memory/max Active (gib): 137.59 +- Memory/max Allocated (gib): 137.59 +- Memory/device Reserved (gib): 139.39 + +## Model description + +More information needed + +## Intended uses & limitations + +More information needed + +## Training and evaluation data + +More information needed + +## Training procedure + +### Training hyperparameters + +The following hyperparameters were used during training: +- learning_rate: 4e-05 +- train_batch_size: 2 +- eval_batch_size: 2 +- seed: 42 +- gradient_accumulation_steps: 8 +- total_train_batch_size: 16 +- optimizer: Use OptimizerNames.ADAMW_8BIT with betas=(0.9,0.999) and epsilon=1e-08 and optimizer_args=No additional optimizer arguments +- lr_scheduler_type: constant_with_warmup +- lr_scheduler_warmup_steps: 467 +- training_steps: 46749 + +### Training results + +| Training Loss | Epoch | Step | Validation Loss | Ppl | Active (gib) | Allocated (gib) | Reserved (gib) | +|:-------------:|:------:|:-----:|:---------------:|:------:|:------------:|:---------------:|:--------------:| +| No log | 0 | 0 | 2.0274 | 7.5945 | 33.35 | 33.35 | 33.44 | +| 1.9949 | 0.0214 | 1000 | 1.9929 | 7.3368 | 65.62 | 65.62 | 139.36 | +| 1.9786 | 0.0428 | 2000 | 1.9805 | 7.2465 | 65.62 | 65.62 | 139.39 | +| 1.9634 | 0.0642 | 3000 | 1.9727 | 7.1902 | 65.62 | 65.62 | 139.39 | +| 1.9636 | 0.0856 | 4000 | 1.9664 | 7.1452 | 65.62 | 65.62 | 139.39 | +| 1.9498 | 0.1069 | 5000 | 1.9614 | 7.1090 | 65.62 | 65.62 | 139.39 | +| 1.9595 | 0.1283 | 6000 | 1.9569 | 7.0775 | 65.62 | 65.62 | 139.49 | +| 1.9473 | 0.1497 | 7000 | 1.9514 | 7.0387 | 65.62 | 65.62 | 139.39 | +| 1.9472 | 0.1711 | 8000 | 1.9498 | 7.0273 | 65.62 | 65.62 | 139.39 | +| 1.9413 | 0.1925 | 9000 | 1.9471 | 7.0086 | 65.62 | 65.62 | 139.39 | +| 1.9363 | 0.2139 | 10000 | 1.9445 | 6.9900 | 65.62 | 65.62 | 139.39 | +| 1.9367 | 0.2353 | 11000 | 1.9371 | 6.9383 | 65.62 | 65.62 | 139.49 | +| 1.9306 | 0.2567 | 12000 | 1.9346 | 6.9213 | 65.62 | 65.62 | 139.39 | +| 1.9305 | 0.2781 | 13000 | 1.9323 | 6.9054 | 65.62 | 65.62 | 139.39 | +| 1.9193 | 0.2995 | 14000 | 1.9301 | 6.8899 | 65.62 | 65.62 | 139.39 | +| 1.9237 | 0.3208 | 15000 | 1.9281 | 6.8764 | 65.62 | 65.62 | 139.39 | +| 1.9206 | 0.3422 | 16000 | 1.9269 | 6.8685 | 65.62 | 65.62 | 139.49 | +| 1.9254 | 0.3636 | 17000 | 1.9246 | 6.8524 | 65.62 | 65.62 | 139.39 | +| 1.9142 | 0.3850 | 18000 | 1.9218 | 6.8334 | 65.62 | 65.62 | 139.39 | +| 1.9212 | 0.4064 | 19000 | 1.9209 | 6.8270 | 65.62 | 65.62 | 139.39 | +| 1.9099 | 0.4278 | 20000 | 1.9188 | 6.8129 | 65.62 | 65.62 | 139.39 | +| 1.9175 | 0.4492 | 21000 | 1.9174 | 6.8031 | 65.62 | 65.62 | 139.49 | +| 1.9009 | 0.4706 | 22000 | 1.9177 | 6.8050 | 65.62 | 65.62 | 139.39 | +| 1.9144 | 0.4920 | 23000 | 1.9151 | 6.7873 | 65.62 | 65.62 | 139.39 | +| 1.9087 | 0.5134 | 24000 | 1.9125 | 6.7699 | 65.62 | 65.62 | 139.39 | +| 1.9039 | 0.5347 | 25000 | 1.9122 | 6.7682 | 65.62 | 65.62 | 139.39 | +| 1.8991 | 0.5561 | 26000 | 1.9096 | 6.7504 | 65.62 | 65.62 | 139.49 | +| 1.8998 | 0.5775 | 27000 | 1.9097 | 6.7513 | 65.62 | 65.62 | 139.39 | +| 1.9083 | 0.5989 | 28000 | 1.9077 | 6.7376 | 65.62 | 65.62 | 139.39 | +| 1.9060 | 0.6203 | 29000 | 1.9067 | 6.731 | 65.62 | 65.62 | 139.39 | +| 1.9066 | 0.6417 | 30000 | 1.9037 | 6.7106 | 65.62 | 65.62 | 139.39 | +| 1.8953 | 0.6631 | 31000 | 1.9025 | 6.7026 | 65.62 | 65.62 | 139.49 | +| 1.9043 | 0.6845 | 32000 | 1.9026 | 6.7032 | 65.62 | 65.62 | 139.39 | +| 1.8964 | 0.7059 | 33000 | 1.9007 | 6.6908 | 65.62 | 65.62 | 139.39 | +| 1.9115 | 0.7273 | 34000 | 1.9010 | 6.6924 | 65.62 | 65.62 | 139.39 | +| 1.8899 | 0.7486 | 35000 | 1.8997 | 6.6839 | 65.62 | 65.62 | 139.39 | +| 1.8995 | 0.7700 | 36000 | 1.9001 | 6.6867 | 65.62 | 65.62 | 139.49 | +| 1.8852 | 0.7914 | 37000 | 1.8983 | 6.6744 | 65.62 | 65.62 | 139.39 | +| 1.8880 | 0.8128 | 38000 | 1.8981 | 6.6732 | 65.62 | 65.62 | 139.39 | +| 1.8923 | 0.8342 | 39000 | 1.8962 | 6.6603 | 65.62 | 65.62 | 139.39 | +| 1.8901 | 0.8556 | 40000 | 1.8951 | 6.6531 | 65.62 | 65.62 | 139.39 | +| 1.8925 | 0.8770 | 41000 | 1.8964 | 6.6620 | 65.62 | 65.62 | 139.49 | +| 1.8947 | 0.8984 | 42000 | 1.8936 | 6.6435 | 65.62 | 65.62 | 139.39 | +| 1.8887 | 0.9198 | 43000 | 1.8942 | 6.6471 | 65.62 | 65.62 | 139.39 | +| 1.8871 | 0.9412 | 44000 | 1.8934 | 6.6419 | 65.62 | 65.62 | 139.39 | +| 1.8853 | 0.9625 | 45000 | 1.8932 | 6.6405 | 65.62 | 65.62 | 139.39 | +| 1.8899 | 0.9839 | 46000 | 1.8913 | 6.6281 | 65.62 | 65.62 | 139.49 | +| 1.8791 | 1.0000 | 46749 | 1.8920 | 6.6325 | 137.59 | 137.59 | 139.39 | + + +### Framework versions + +- Transformers 5.9.0 +- Pytorch 2.10.0+cu128 +- Datasets 4.8.5 +- Tokenizers 0.22.2 diff --git a/axolotl_config.yml b/axolotl_config.yml new file mode 100644 index 0000000..d476350 --- /dev/null +++ b/axolotl_config.yml @@ -0,0 +1,66 @@ +base_model: Qwen/Qwen3-14B-Base + +trust_remote_code: true +tokenizer_use_fast: true + +load_in_8bit: false +load_in_4bit: false + + +datasets: + - path: philipperen55/datasetCPT70axolotlRandomized + split: train + data_files: datasetCPT70axolotlRandomized.jsonl + type: completion + field: text + + +val_set_size: 0.0001 + +dataset_prepared_path: prepared_cpt +output_dir: outputs_cpt + +sequence_len: 2048 +pad_to_sequence_len: true +sample_packing: true +eval_sample_packing: true +attn_implementation: flash_attention_2 +excess_length_strategy: truncate +train_on_inputs: true +add_eos_token: true + +# FULL finetune CPT +micro_batch_size: 2 +gradient_accumulation_steps: 8 +num_epochs: 1 + + +optimizer: adamw_8bit +learning_rate: 4e-5 +weight_decay: 0.01 +lr_scheduler: constant_with_warmup +warmup_ratio: 0.01 +max_grad_norm: 1.0 + +fp16: false +bf16: true +tf32: true +gradient_checkpointing: false + +logging_steps: 50 +eval_steps: 1000 +save_steps: 5000 +save_total_limit: 1 +save_only_model: true + +seed: 42 +#mettre 24 si ya plus de 24 vspu, sinon mettre 16 si ya 24vcpu +dataset_num_proc: 24 + +# WandB +wandb_project: qwen3_14b_cpt_full_axolotl + +# Hub +hub_model_id: +push_to_hub: false #mon script upload de façon plus sûre +hub_strategy: every_save diff --git a/config.json b/config.json new file mode 100644 index 0000000..44cd204 --- /dev/null +++ b/config.json @@ -0,0 +1,75 @@ +{ + "architectures": [ + "Qwen3ForCausalLM" + ], + "attention_bias": false, + "attention_dropout": 0.0, + "bos_token_id": null, + "dtype": "bfloat16", + "eos_token_id": 151643, + "head_dim": 128, + "hidden_act": "silu", + "hidden_size": 5120, + "initializer_range": 0.02, + "intermediate_size": 17408, + "layer_types": [ + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention" + ], + "max_position_embeddings": 32768, + "max_window_layers": 40, + "model_type": "qwen3", + "num_attention_heads": 40, + "num_hidden_layers": 40, + "num_key_value_heads": 8, + "pad_token_id": 151643, + "rms_norm_eps": 1e-06, + "rope_parameters": { + "rope_theta": 1000000, + "rope_type": "default" + }, + "sliding_window": null, + "tie_word_embeddings": false, + "transformers_version": "5.9.0", + "use_cache": false, + "use_sliding_window": false, + "vocab_size": 151936 +} diff --git a/generation_config.json b/generation_config.json new file mode 100644 index 0000000..892f08a --- /dev/null +++ b/generation_config.json @@ -0,0 +1,9 @@ +{ + "do_sample": true, + "eos_token_id": [ + 151643 + ], + "max_new_tokens": 2048, + "pad_token_id": 151643, + "transformers_version": "5.9.0" +} diff --git a/model.safetensors b/model.safetensors new file mode 100644 index 0000000..5040569 --- /dev/null +++ b/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ac4383d4f573d1b04579c0cdaed3be526fac58b8ff87578952dfd42e76ed164b +size 29536666272 diff --git a/tokenizer.json b/tokenizer.json new file mode 100644 index 0000000..c7afbed --- /dev/null +++ b/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:be75606093db2094d7cd20f3c2f385c212750648bd6ea4fb2bf507a6a4c55506 +size 11422650 diff --git a/tokenizer_config.json b/tokenizer_config.json new file mode 100644 index 0000000..6359c62 --- /dev/null +++ b/tokenizer_config.json @@ -0,0 +1,31 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": {}, + "is_local": false, + "local_files_only": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null, + "additional_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ] +} diff --git a/training_args.bin b/training_args.bin new file mode 100644 index 0000000..721b9ee --- /dev/null +++ b/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8c68270c07a85de3badf23f548b80177a4429ea04fed17bd0335b511cbb67ce4 +size 6737