commit 64305eee086807813bdf144cfbce17fa70844fe4 Author: ModelHub XC Date: Mon Jul 6 20:09:13 2026 +0800 初始化项目,由ModelHub XC社区提供模型 Model: QuantFactory/Einstein-v6.1-Llama3-8B-GGUF Source: Original Platform diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..3323f66 --- /dev/null +++ b/.gitattributes @@ -0,0 +1,49 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text +Einstein-v6.1-Llama3-8B.Q2_K.gguf filter=lfs diff=lfs merge=lfs -text +Einstein-v6.1-Llama3-8B.Q3_K_L.gguf filter=lfs diff=lfs merge=lfs -text +Einstein-v6.1-Llama3-8B.Q3_K_M.gguf filter=lfs diff=lfs merge=lfs -text +Einstein-v6.1-Llama3-8B.Q3_K_S.gguf filter=lfs diff=lfs merge=lfs -text +Einstein-v6.1-Llama3-8B.Q4_0.gguf filter=lfs diff=lfs merge=lfs -text +Einstein-v6.1-Llama3-8B.Q4_1.gguf filter=lfs diff=lfs merge=lfs -text +Einstein-v6.1-Llama3-8B.Q4_K_M.gguf filter=lfs diff=lfs merge=lfs -text +Einstein-v6.1-Llama3-8B.Q4_K_S.gguf filter=lfs diff=lfs merge=lfs -text +Einstein-v6.1-Llama3-8B.Q5_0.gguf filter=lfs diff=lfs merge=lfs -text +Einstein-v6.1-Llama3-8B.Q5_1.gguf filter=lfs diff=lfs merge=lfs -text +Einstein-v6.1-Llama3-8B.Q5_K_M.gguf filter=lfs diff=lfs merge=lfs -text +Einstein-v6.1-Llama3-8B.Q5_K_S.gguf filter=lfs diff=lfs merge=lfs -text +Einstein-v6.1-Llama3-8B.Q6_K.gguf filter=lfs diff=lfs merge=lfs -text +Einstein-v6.1-Llama3-8B.Q8_0.gguf filter=lfs diff=lfs merge=lfs -text diff --git a/Einstein-v6.1-Llama3-8B.Q2_K.gguf b/Einstein-v6.1-Llama3-8B.Q2_K.gguf new file mode 100644 index 0000000..f28256d --- /dev/null +++ b/Einstein-v6.1-Llama3-8B.Q2_K.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:eeab94efb262e4b2ef5ed23c623b462baf553b077dabd80e2fa7afb95b6214d2 +size 3179152000 diff --git a/Einstein-v6.1-Llama3-8B.Q3_K_L.gguf b/Einstein-v6.1-Llama3-8B.Q3_K_L.gguf new file mode 100644 index 0000000..aa1b478 --- /dev/null +++ b/Einstein-v6.1-Llama3-8B.Q3_K_L.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:06fc7e3e7e8fcfa87a7e21876ed48799978633a9a7b59fab483ad5649145923f +size 4321978624 diff --git a/Einstein-v6.1-Llama3-8B.Q3_K_M.gguf b/Einstein-v6.1-Llama3-8B.Q3_K_M.gguf new file mode 100644 index 0000000..3956969 --- /dev/null +++ b/Einstein-v6.1-Llama3-8B.Q3_K_M.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:981bc5f8362457fecebc55dd9a439885ce171ee1892ebff1d926ef1be4b925eb +size 4018940160 diff --git a/Einstein-v6.1-Llama3-8B.Q3_K_S.gguf b/Einstein-v6.1-Llama3-8B.Q3_K_S.gguf new file mode 100644 index 0000000..865065d --- /dev/null +++ b/Einstein-v6.1-Llama3-8B.Q3_K_S.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:36d126615e811458648f3aa13b9134e76721b4fd936b0d5b459ccdcd7c5049fc +size 3664521472 diff --git a/Einstein-v6.1-Llama3-8B.Q4_0.gguf b/Einstein-v6.1-Llama3-8B.Q4_0.gguf new file mode 100644 index 0000000..cde2d15 --- /dev/null +++ b/Einstein-v6.1-Llama3-8B.Q4_0.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3481488571ac890043b374da08d1428cb2c576606b45e87a89e5a84feda19db4 +size 4661236096 diff --git a/Einstein-v6.1-Llama3-8B.Q4_1.gguf b/Einstein-v6.1-Llama3-8B.Q4_1.gguf new file mode 100644 index 0000000..f592601 --- /dev/null +++ b/Einstein-v6.1-Llama3-8B.Q4_1.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d0252176a00ba50b719477f71c0adf12fe5b952f781924a82356ae1dae3fbbc4 +size 5130278272 diff --git a/Einstein-v6.1-Llama3-8B.Q4_K_M.gguf b/Einstein-v6.1-Llama3-8B.Q4_K_M.gguf new file mode 100644 index 0000000..ccb61e5 --- /dev/null +++ b/Einstein-v6.1-Llama3-8B.Q4_K_M.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c4191ae5628582e46fca1a1faf35b199243cf656b2e60b4ddc15e408a4b49e3f +size 4920758656 diff --git a/Einstein-v6.1-Llama3-8B.Q4_K_S.gguf b/Einstein-v6.1-Llama3-8B.Q4_K_S.gguf new file mode 100644 index 0000000..4833950 --- /dev/null +++ b/Einstein-v6.1-Llama3-8B.Q4_K_S.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2f541a702ed9434e10f4720994342000c34d9e2edfb091ce6e464ba20f128d20 +size 4692693376 diff --git a/Einstein-v6.1-Llama3-8B.Q5_0.gguf b/Einstein-v6.1-Llama3-8B.Q5_0.gguf new file mode 100644 index 0000000..c64b491 --- /dev/null +++ b/Einstein-v6.1-Llama3-8B.Q5_0.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d30a82ba14e24be5818fe1e54a40ad8ce2bfaab90203ebe85f3f4083713bfb0b +size 5599320448 diff --git a/Einstein-v6.1-Llama3-8B.Q5_1.gguf b/Einstein-v6.1-Llama3-8B.Q5_1.gguf new file mode 100644 index 0000000..b1ad5ad --- /dev/null +++ b/Einstein-v6.1-Llama3-8B.Q5_1.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a5fdc9fb1ba3abae4a9200bacd6f09c15b49c0a28b458f6859f475ac18ffb440 +size 6068362624 diff --git a/Einstein-v6.1-Llama3-8B.Q5_K_M.gguf b/Einstein-v6.1-Llama3-8B.Q5_K_M.gguf new file mode 100644 index 0000000..603e39c --- /dev/null +++ b/Einstein-v6.1-Llama3-8B.Q5_K_M.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2fa1f74c3082e7b31039e64476d36c6b54058141e70fc5f043fc478c27d36908 +size 5733013888 diff --git a/Einstein-v6.1-Llama3-8B.Q5_K_S.gguf b/Einstein-v6.1-Llama3-8B.Q5_K_S.gguf new file mode 100644 index 0000000..edd2e96 --- /dev/null +++ b/Einstein-v6.1-Llama3-8B.Q5_K_S.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:add41a4e7e6f4cceced3f67c230b8ae6a6c5e108078e12578f1ed9ac47e10aef +size 5599320448 diff --git a/Einstein-v6.1-Llama3-8B.Q6_K.gguf b/Einstein-v6.1-Llama3-8B.Q6_K.gguf new file mode 100644 index 0000000..2b480ff --- /dev/null +++ b/Einstein-v6.1-Llama3-8B.Q6_K.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:75725574670315f765a8b4b029fa82c5adedb823e78bac24f7790b1fa0fccb19 +size 6596035072 diff --git a/Einstein-v6.1-Llama3-8B.Q8_0.gguf b/Einstein-v6.1-Llama3-8B.Q8_0.gguf new file mode 100644 index 0000000..0526297 --- /dev/null +++ b/Einstein-v6.1-Llama3-8B.Q8_0.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:490308051f32bd5262cbcb65e0594c30ac12459efa3fb72f315800587f825f84 +size 8540807424 diff --git a/README.md b/README.md new file mode 100644 index 0000000..6cd9a81 --- /dev/null +++ b/README.md @@ -0,0 +1,554 @@ + +--- + +language: +- en +license: other +tags: +- axolotl +- generated_from_trainer +- instruct +- finetune +- chatml +- gpt4 +- synthetic data +- science +- physics +- chemistry +- biology +- math +- llama +- llama3 +base_model: meta-llama/Meta-Llama-3-8B +datasets: +- allenai/ai2_arc +- camel-ai/physics +- camel-ai/chemistry +- camel-ai/biology +- camel-ai/math +- metaeval/reclor +- openbookqa +- mandyyyyii/scibench +- derek-thomas/ScienceQA +- TIGER-Lab/ScienceEval +- jondurbin/airoboros-3.2 +- LDJnr/Capybara +- Cot-Alpaca-GPT4-From-OpenHermes-2.5 +- STEM-AI-mtl/Electrical-engineering +- knowrohit07/saraswati-stem +- sablo/oasst2_curated +- lmsys/lmsys-chat-1m +- TIGER-Lab/MathInstruct +- bigbio/med_qa +- meta-math/MetaMathQA-40K +- openbookqa +- piqa +- metaeval/reclor +- derek-thomas/ScienceQA +- scibench +- sciq +- Open-Orca/SlimOrca +- migtissera/Synthia-v1.3 +- TIGER-Lab/ScienceEval +- allenai/WildChat +- microsoft/orca-math-word-problems-200k +- openchat/openchat_sharegpt4_dataset +- teknium/GPTeacher-General-Instruct +- m-a-p/CodeFeedback-Filtered-Instruction +- totally-not-an-llm/EverythingLM-data-V3 +- HuggingFaceH4/no_robots +- OpenAssistant/oasst_top1_2023-08-25 +- WizardLM/WizardLM_evol_instruct_70k +model-index: +- name: Einstein-v6.1-Llama3-8B + results: + - task: + type: text-generation + name: Text Generation + dataset: + name: AI2 Reasoning Challenge (25-Shot) + type: ai2_arc + config: ARC-Challenge + split: test + args: + num_few_shot: 25 + metrics: + - type: acc_norm + value: 62.46 + name: normalized accuracy + source: + url: https://huggingface.co/spaces/HuggingFaceH4/open_llm_leaderboard?query=Weyaxi/Einstein-v6.1-Llama3-8B + name: Open LLM Leaderboard + - task: + type: text-generation + name: Text Generation + dataset: + name: HellaSwag (10-Shot) + type: hellaswag + split: validation + args: + num_few_shot: 10 + metrics: + - type: acc_norm + value: 82.41 + name: normalized accuracy + source: + url: https://huggingface.co/spaces/HuggingFaceH4/open_llm_leaderboard?query=Weyaxi/Einstein-v6.1-Llama3-8B + name: Open LLM Leaderboard + - task: + type: text-generation + name: Text Generation + dataset: + name: MMLU (5-Shot) + type: cais/mmlu + config: all + split: test + args: + num_few_shot: 5 + metrics: + - type: acc + value: 66.19 + name: accuracy + source: + url: https://huggingface.co/spaces/HuggingFaceH4/open_llm_leaderboard?query=Weyaxi/Einstein-v6.1-Llama3-8B + name: Open LLM Leaderboard + - task: + type: text-generation + name: Text Generation + dataset: + name: TruthfulQA (0-shot) + type: truthful_qa + config: multiple_choice + split: validation + args: + num_few_shot: 0 + metrics: + - type: mc2 + value: 55.1 + source: + url: https://huggingface.co/spaces/HuggingFaceH4/open_llm_leaderboard?query=Weyaxi/Einstein-v6.1-Llama3-8B + name: Open LLM Leaderboard + - task: + type: text-generation + name: Text Generation + dataset: + name: Winogrande (5-shot) + type: winogrande + config: winogrande_xl + split: validation + args: + num_few_shot: 5 + metrics: + - type: acc + value: 79.32 + name: accuracy + source: + url: https://huggingface.co/spaces/HuggingFaceH4/open_llm_leaderboard?query=Weyaxi/Einstein-v6.1-Llama3-8B + name: Open LLM Leaderboard + - task: + type: text-generation + name: Text Generation + dataset: + name: GSM8k (5-shot) + type: gsm8k + config: main + split: test + args: + num_few_shot: 5 + metrics: + - type: acc + value: 66.11 + name: accuracy + source: + url: https://huggingface.co/spaces/HuggingFaceH4/open_llm_leaderboard?query=Weyaxi/Einstein-v6.1-Llama3-8B + name: Open LLM Leaderboard + - task: + type: text-generation + name: Text Generation + dataset: + name: IFEval (0-Shot) + type: HuggingFaceH4/ifeval + args: + num_few_shot: 0 + metrics: + - type: inst_level_strict_acc and prompt_level_strict_acc + value: 45.68 + name: strict accuracy + source: + url: https://huggingface.co/spaces/open-llm-leaderboard/open_llm_leaderboard?query=Weyaxi/Einstein-v6.1-Llama3-8B + name: Open LLM Leaderboard + - task: + type: text-generation + name: Text Generation + dataset: + name: BBH (3-Shot) + type: BBH + args: + num_few_shot: 3 + metrics: + - type: acc_norm + value: 29.38 + name: normalized accuracy + source: + url: https://huggingface.co/spaces/open-llm-leaderboard/open_llm_leaderboard?query=Weyaxi/Einstein-v6.1-Llama3-8B + name: Open LLM Leaderboard + - task: + type: text-generation + name: Text Generation + dataset: + name: MATH Lvl 5 (4-Shot) + type: hendrycks/competition_math + args: + num_few_shot: 4 + metrics: + - type: exact_match + value: 5.74 + name: exact match + source: + url: https://huggingface.co/spaces/open-llm-leaderboard/open_llm_leaderboard?query=Weyaxi/Einstein-v6.1-Llama3-8B + name: Open LLM Leaderboard + - task: + type: text-generation + name: Text Generation + dataset: + name: GPQA (0-shot) + type: Idavidrein/gpqa + args: + num_few_shot: 0 + metrics: + - type: acc_norm + value: 4.25 + name: acc_norm + source: + url: https://huggingface.co/spaces/open-llm-leaderboard/open_llm_leaderboard?query=Weyaxi/Einstein-v6.1-Llama3-8B + name: Open LLM Leaderboard + - task: + type: text-generation + name: Text Generation + dataset: + name: MuSR (0-shot) + type: TAUR-Lab/MuSR + args: + num_few_shot: 0 + metrics: + - type: acc_norm + value: 11.23 + name: acc_norm + source: + url: https://huggingface.co/spaces/open-llm-leaderboard/open_llm_leaderboard?query=Weyaxi/Einstein-v6.1-Llama3-8B + name: Open LLM Leaderboard + - task: + type: text-generation + name: Text Generation + dataset: + name: MMLU-PRO (5-shot) + type: TIGER-Lab/MMLU-Pro + config: main + split: test + args: + num_few_shot: 5 + metrics: + - type: acc + value: 23.68 + name: accuracy + source: + url: https://huggingface.co/spaces/open-llm-leaderboard/open_llm_leaderboard?query=Weyaxi/Einstein-v6.1-Llama3-8B + name: Open LLM Leaderboard + +--- + +[![QuantFactory Banner](https://lh7-rt.googleusercontent.com/docsz/AD_4nXeiuCm7c8lEwEJuRey9kiVZsRn2W-b4pWlu3-X534V3YmVuVc2ZL-NXg2RkzSOOS2JXGHutDuyyNAUtdJI65jGTo8jT9Y99tMi4H4MqL44Uc5QKG77B0d6-JfIkZHFaUA71-RtjyYZWVIhqsNZcx8-OMaA?key=xt3VSDoCbmTY7o-cwwOFwQ)](https://hf.co/QuantFactory) + + +# QuantFactory/Einstein-v6.1-Llama3-8B-GGUF +This is quantized version of [Weyaxi/Einstein-v6.1-Llama3-8B](https://huggingface.co/Weyaxi/Einstein-v6.1-Llama3-8B) created using llama.cpp + +# Original Model Card + +![image/png](https://cdn-uploads.huggingface.co/production/uploads/6468ce47e134d050a58aa89c/5s12oq859qLfDkkTNam_C.png) + +# 🔬 Einstein-v6.1-Llama3-8B + +This model is a full fine-tuned version of [meta-llama/Meta-Llama-3-8B](https://huggingface.co/meta-llama/Meta-Llama-3-8B) on diverse datasets. + +This model is finetuned using `8xRTX3090` + `1xRTXA6000` using [axolotl](https://github.com/OpenAccess-AI-Collective/axolotl). + +This model's training was sponsored by [sablo.ai](https://sablo.ai). + +
See axolotl config + +axolotl version: `0.4.0` +```yaml +base_model: meta-llama/Meta-Llama-3-8B +model_type: LlamaForCausalLM +tokenizer_type: AutoTokenizer + +load_in_8bit: false +load_in_4bit: false +strict: false + +chat_template: chatml +datasets: + - path: data/merged_all.json + ds_type: json + type: alpaca + conversation: chatml + + - path: data/gpteacher-instruct-special-alpaca.json + ds_type: json + type: gpteacher + conversation: chatml + + - path: data/wizardlm_evol_instruct_70k_random_half.json + ds_type: json + type: alpaca + conversation: chatml + + - path: data/capybara_sharegpt.json + ds_type: json + type: sharegpt + conversation: chatml + + - path: data/synthia-v1.3_sharegpt_12500.json + ds_type: json + type: sharegpt + conversation: chatml + + - path: data/cot_alpaca_gpt4_extracted_openhermes_2.5_sharegpt.json + ds_type: json + type: sharegpt + conversation: chatml + + - path: data/slimorca_dedup_filtered_95k_sharegpt.json + ds_type: json + type: sharegpt + conversation: chatml + + - path: data/airoboros_3.2_without_contextual_slimorca_orca_sharegpt.json + ds_type: json + type: sharegpt + conversation: chatml + + - path: data/allenai_wild_chat_gpt4_english_toxic_random_half_4k_sharegpt.json + ds_type: json + type: sharegpt + strict: false + conversation: chatml + + - path: data/pippa_bagel_repo_3k_sharegpt.json + ds_type: json + type: sharegpt + conversation: chatml + + - path: data/gpt4_data_lmys_1m_sharegpt.json + ds_type: json + type: sharegpt + conversation: chatml + + - path: data/sharegpt_gpt4_english.json + ds_type: json + type: sharegpt + conversation: chatml + + - path: data/no_robots_sharegpt.json + ds_type: json + type: sharegpt + strict: false + conversation: chatml + + - path: data/oasst_top1_from_fusechatmixture_sharegpt.json + ds_type: json + type: sharegpt + strict: false + conversation: chatml + + - path: data/everythinglm-data-v3_sharegpt.json + ds_type: json + type: sharegpt + strict: false + conversation: chatml + +dataset_prepared_path: last_run_prepared +val_set_size: 0.002 + +output_dir: ./Einstein-v6.1-Llama3-8B-model + +sequence_len: 8192 +sample_packing: true +pad_to_sequence_len: true +eval_sample_packing: false + +wandb_project: Einstein +wandb_entity: +wandb_watch: +wandb_name: Einstein-v6.1-Llama3-2-epoch +wandb_log_model: +hub_model_id: Weyaxi/Einstein-v6.1-Llama3-8B + +save_safetensors: true + +gradient_accumulation_steps: 4 +micro_batch_size: 1 +num_epochs: 2 +optimizer: adamw_bnb_8bit # look +lr_scheduler: cosine +learning_rate: 0.000005 # look + +train_on_inputs: false +group_by_length: false +bf16: true +fp16: false +tf32: false + +gradient_checkpointing: true +early_stopping_patience: +resume_from_checkpoint: +local_rank: +logging_steps: 1 +xformers_attention: +flash_attention: true + +warmup_steps: 10 +evals_per_epoch: 2 +eval_table_size: +eval_table_max_new_tokens: 128 +saves_per_epoch: 2 +debug: + +deepspeed: zero3_bf16_cpuoffload_params.json +weight_decay: 0.0 +fsdp: +fsdp_config: +special_tokens: + bos_token: "" + eos_token: "<|im_end|>" + unk_token: "" + pad_token: <|end_of_text|> # changed +tokens: + - "<|im_start|>" +``` +

+ +# 💬 Prompt Template + +You can use ChatML prompt template while using the model: + +### ChatML + +``` +<|im_start|>system +{system}<|im_end|> +<|im_start|>user +{user}<|im_end|> +<|im_start|>assistant +{asistant}<|im_end|> +``` + +This prompt template is available as a [chat template](https://huggingface.co/docs/transformers/main/chat_templating), which means you can format messages using the +`tokenizer.apply_chat_template()` method: + +```python +messages = [ + {"role": "system", "content": "You are helpful AI asistant."}, + {"role": "user", "content": "Hello!"} +] +gen_input = tokenizer.apply_chat_template(message, return_tensors="pt") +model.generate(**gen_input) +``` + +# 📊 Datasets used in this model + +The datasets used to train this model are listed in the metadata section of the model card. + +Please note that certain datasets mentioned in the metadata may have undergone filtering based on various criteria. + +The results of this filtering process and its outcomes are in the data folder of this repository: + +[Weyaxi/Einstein-v6.1-Llama3-8B/data](https://huggingface.co/Weyaxi/Einstein-v6.1-Llama3-8B/tree/main/data) + +# 🔄 Quantizationed versions + +## GGUF [@bartowski](https://huggingface.co/bartowski) + +- https://huggingface.co/bartowski/Einstein-v6.1-Llama3-8B-GGUF + +## ExLlamaV2 [@bartowski](https://huggingface.co/bartowski) + +- https://huggingface.co/bartowski/Einstein-v6.1-Llama3-8B-exl2 + +## AWQ [@solidrust](https://huggingface.co/solidrust) + +- https://huggingface.co/solidrust/Einstein-v6.1-Llama3-8B-AWQ + +# 🎯 [Open LLM Leaderboard Evaluation Results](https://huggingface.co/spaces/HuggingFaceH4/open_llm_leaderboard) +Detailed results can be found [here](https://huggingface.co/datasets/open-llm-leaderboard/details_Weyaxi__Einstein-v6.1-Llama3-8B) + +| Metric |Value| +|---------------------------------|----:| +|Avg. |68.60| +|AI2 Reasoning Challenge (25-Shot)|62.46| +|HellaSwag (10-Shot) |82.41| +|MMLU (5-Shot) |66.19| +|TruthfulQA (0-shot) |55.10| +|Winogrande (5-shot) |79.32| +|GSM8k (5-shot) |66.11| + +# 🎯 [Open LLM Leaderboard v2 Evaluation Results](https://huggingface.co/spaces/open-llm-leaderboard/open_llm_leaderboard) +Detailed results can be found [here](https://huggingface.co/datasets/open-llm-leaderboard/details_Weyaxi__Einstein-v6.1-Llama3-8B) + +| Metric |Value| +|-------------------|----:| +|Avg. |19.99| +|IFEval (0-Shot) |45.68| +|BBH (3-Shot) |29.38| +|MATH Lvl 5 (4-Shot)| 5.74| +|GPQA (0-shot) | 4.25| +|MuSR (0-shot) |11.23| +|MMLU-PRO (5-shot) |23.68| + + +# 📚 Some resources, discussions and reviews aboout this model + +#### 🐦 Announcement tweet: + +- https://twitter.com/Weyaxi/status/1783050724659675627 + +#### 🔍 Reddit post in r/LocalLLaMA: + +- https://www.reddit.com/r/LocalLLaMA/comments/1cdlym1/introducing_einstein_v61_based_on_the_new_llama3/ + +#### ▶️ Youtube Video(s) + +- [Install Einstein v6.1 Llama3-8B Locally on Windows](https://www.youtube.com/watch?v=VePvv6OM0JY) + +#### 📱 Octopus-V4-3B + +- [Octopus-V4-3B](https://huggingface.co/NexaAIDev/Octopus-v4) leverages the incredible physics capabilities of [Einstein-v6.1-Llama3-8B](https://huggingface.co/Weyaxi/Einstein-v6.1-Llama3-8B) in their model. + +# 🤖 Additional information about training + +This model is full fine-tuned for 2 epoch. + +Total number of steps was 2026. + +
Loss graph + +![image/png](https://cdn-uploads.huggingface.co/production/uploads/6468ce47e134d050a58aa89c/Ycs7ZpoqmxFt0u9rybCO1.png) + +

+ +# 🤝 Acknowledgments + +Thanks to [sablo.ai](https://sablo.ai) for sponsoring this model. + +Thanks to all the dataset authors mentioned in the datasets section. + +Thanks to [axolotl](https://github.com/OpenAccess-AI-Collective/axolotl) for making the repository I used to make this model. + +Thanks to all open source AI community. + +[Built with Axolotl](https://github.com/OpenAccess-AI-Collective/axolotl) + +If you would like to support me: + +[☕ Buy Me a Coffee](https://www.buymeacoffee.com/weyaxi) diff --git a/configuration.json b/configuration.json new file mode 100644 index 0000000..bbeeda1 --- /dev/null +++ b/configuration.json @@ -0,0 +1 @@ +{"framework": "pytorch", "task": "text-generation", "allow_remote": true} \ No newline at end of file