初始化项目,由ModelHub XC社区提供模型

Model: ThomasBaruzier/EXAONE-3.5-7.8B-Instruct-GGUF
Source: Original Platform
This commit is contained in:
ModelHub XC
2026-09-12 20:04:22 +08:00
commit d3e8f80624
32 changed files with 426 additions and 0 deletions

64
.gitattributes vendored Normal file
View File

@@ -0,0 +1,64 @@
*.7z filter=lfs diff=lfs merge=lfs -text
*.arrow filter=lfs diff=lfs merge=lfs -text
*.bin filter=lfs diff=lfs merge=lfs -text
*.bz2 filter=lfs diff=lfs merge=lfs -text
*.ckpt filter=lfs diff=lfs merge=lfs -text
*.ftz filter=lfs diff=lfs merge=lfs -text
*.gz filter=lfs diff=lfs merge=lfs -text
*.h5 filter=lfs diff=lfs merge=lfs -text
*.joblib filter=lfs diff=lfs merge=lfs -text
*.lfs.* filter=lfs diff=lfs merge=lfs -text
*.mlmodel filter=lfs diff=lfs merge=lfs -text
*.model filter=lfs diff=lfs merge=lfs -text
*.msgpack filter=lfs diff=lfs merge=lfs -text
*.npy filter=lfs diff=lfs merge=lfs -text
*.npz filter=lfs diff=lfs merge=lfs -text
*.onnx filter=lfs diff=lfs merge=lfs -text
*.ot filter=lfs diff=lfs merge=lfs -text
*.parquet filter=lfs diff=lfs merge=lfs -text
*.pb filter=lfs diff=lfs merge=lfs -text
*.pickle filter=lfs diff=lfs merge=lfs -text
*.pkl filter=lfs diff=lfs merge=lfs -text
*.pt filter=lfs diff=lfs merge=lfs -text
*.pth filter=lfs diff=lfs merge=lfs -text
*.rar filter=lfs diff=lfs merge=lfs -text
*.safetensors filter=lfs diff=lfs merge=lfs -text
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
*.tar.* filter=lfs diff=lfs merge=lfs -text
*.tar filter=lfs diff=lfs merge=lfs -text
*.tflite filter=lfs diff=lfs merge=lfs -text
*.tgz filter=lfs diff=lfs merge=lfs -text
*.wasm filter=lfs diff=lfs merge=lfs -text
*.xz filter=lfs diff=lfs merge=lfs -text
*.zip filter=lfs diff=lfs merge=lfs -text
*.zst filter=lfs diff=lfs merge=lfs -text
*tfevents* filter=lfs diff=lfs merge=lfs -text
imatrix.dat filter=lfs diff=lfs merge=lfs -text
EXAONE-3.5-7.8B-Instruct-IQ1_M.gguf filter=lfs diff=lfs merge=lfs -text
EXAONE-3.5-7.8B-Instruct-IQ1_S.gguf filter=lfs diff=lfs merge=lfs -text
EXAONE-3.5-7.8B-Instruct-Q2_K_S.gguf filter=lfs diff=lfs merge=lfs -text
EXAONE-3.5-7.8B-Instruct-Q2_K.gguf filter=lfs diff=lfs merge=lfs -text
EXAONE-3.5-7.8B-Instruct-IQ2_XXS.gguf filter=lfs diff=lfs merge=lfs -text
EXAONE-3.5-7.8B-Instruct-IQ2_XS.gguf filter=lfs diff=lfs merge=lfs -text
EXAONE-3.5-7.8B-Instruct-IQ2_M.gguf filter=lfs diff=lfs merge=lfs -text
EXAONE-3.5-7.8B-Instruct-IQ2_S.gguf filter=lfs diff=lfs merge=lfs -text
EXAONE-3.5-7.8B-Instruct-Q3_K_S.gguf filter=lfs diff=lfs merge=lfs -text
EXAONE-3.5-7.8B-Instruct-Q3_K_L.gguf filter=lfs diff=lfs merge=lfs -text
EXAONE-3.5-7.8B-Instruct-IQ3_XS.gguf filter=lfs diff=lfs merge=lfs -text
EXAONE-3.5-7.8B-Instruct-IQ3_XXS.gguf filter=lfs diff=lfs merge=lfs -text
EXAONE-3.5-7.8B-Instruct-Q3_K_M.gguf filter=lfs diff=lfs merge=lfs -text
EXAONE-3.5-7.8B-Instruct-IQ3_S.gguf filter=lfs diff=lfs merge=lfs -text
EXAONE-3.5-7.8B-Instruct-IQ3_M.gguf filter=lfs diff=lfs merge=lfs -text
EXAONE-3.5-7.8B-Instruct-Q4_K_M.gguf filter=lfs diff=lfs merge=lfs -text
EXAONE-3.5-7.8B-Instruct-IQ4_XS.gguf filter=lfs diff=lfs merge=lfs -text
EXAONE-3.5-7.8B-Instruct-IQ4_NL.gguf filter=lfs diff=lfs merge=lfs -text
EXAONE-3.5-7.8B-Instruct-Q4_K_S.gguf filter=lfs diff=lfs merge=lfs -text
EXAONE-3.5-7.8B-Instruct-Q5_K_S.gguf filter=lfs diff=lfs merge=lfs -text
EXAONE-3.5-7.8B-Instruct-Q5_K_M.gguf filter=lfs diff=lfs merge=lfs -text
EXAONE-3.5-7.8B-Instruct-Q6_K.gguf filter=lfs diff=lfs merge=lfs -text
EXAONE-3.5-7.8B-Instruct-F16.gguf filter=lfs diff=lfs merge=lfs -text
EXAONE-3.5-7.8B-Instruct-Q4_0.gguf filter=lfs diff=lfs merge=lfs -text
EXAONE-3.5-7.8B-Instruct-Q4_1.gguf filter=lfs diff=lfs merge=lfs -text
EXAONE-3.5-7.8B-Instruct-Q5_0.gguf filter=lfs diff=lfs merge=lfs -text
EXAONE-3.5-7.8B-Instruct-Q5_1.gguf filter=lfs diff=lfs merge=lfs -text
EXAONE-3.5-7.8B-Instruct-Q8_0.gguf filter=lfs diff=lfs merge=lfs -text

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:5c9a65c06234a83316aa40ea490e20139ddb5bce58149c3056ac145a06a19231
size 15641630624

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:70a7d9e0c4856cc8132fe96f1d43f0cf0914d26bec528eb0acea601a8cd54e96
size 2050775232

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:20fe6392fd416d7cf79dcf826c6e9fa6c7d1a84517e9bcbb96fb0e58b4c6eaa2
size 1908431040

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:e2e038f82aaf4e84188591ab9473ba39c615c499c94c038aa6f7d6e6cdf0be79
size 2826328256

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:7c934ed5dd799e9c3e1e3b4a666869ee48e3a7ddeb225ada8a45e6c2a7fb94c9
size 2636536000

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:3ea50f37da53566cb000d2a87839c56b927d7013041d54cc346a5d28b4ba9abc
size 2494585024

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:ba407f83f38896bc771155aa31f1d5bcbfe6288f479f96fd4455e6ad91cd6814
size 2288015552

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:60feea1b8d849644d621b76f458cc05bd8f661786214e78736aeec7ed82c6958
size 3648805056

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:09999f1531cf3aa9d60ea8ffaf287d92df471ac4a80a5fb3b578fe67facc7817
size 3546306752

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:b7b090a6f18f9b99a96593821ed2f57828ccf64ebd919d9bd99e43f87460bc46
size 3382728896

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:5f56e5f649cafb0cacb779517c4d5f9412f7298d794186b7247990bc2d798405
size 3152959680

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:5d72236b0629665049d5baf88f480e57d379e7b80ffac5cbe45fd00b974e3e87
size 4527904960

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:daaa9601b00f1b2924981850e060bf29d691c16c62a75ca5a33a9dcfdadd585f
size 4300888256

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:593c1143ed17a0dfb20a3c843bd07f1aaf6e3408757db273a13c9728a4af7cde
size 3053869248

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:a4b479782b1d90803eadc4d3a542541cfdd0ffca07effcb9b8ee395ed1aaaf8e
size 2863552704

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:f4f81deb6d273ac2a53fc2d813e10d21b43fb761ed303dff8214c306a76b9b8d
size 4185938112

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:d6d7a90bff3d197887d9ad14ef21e394fad4641c3c918b9d2e4a6965c8ea6be0
size 3882899648

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:0920b3d56e649d58289be0bfa4d7500c63097e2c84dd1a34ec42aa6526edeb29
size 3528480960

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:f9e352b60bc9ad0855a48be54c60517bc6f36b1d02b3e027a34b8095736f355e
size 4525807808

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:cc214e07e3ee2b0407a2b137b63181ad05baea07285eb94b98fdeb78bcdca918
size 4973549760

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:efa9d654681ea40c74d8494952601add9e69a11276ff2720e15f9885d68f4ad8
size 4770650304

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:d2c25cfb94f18f9a73a8b50cae03030bfd0a189967ae867212685fcd61a4c8cf
size 4542585024

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:d7858b81ff8991c3aa6a1c6127da44a31607ae01e73f4ca35e7450a089f800cd
size 5450651840

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:10e0ad27cd01e806e348b19274ec4c7dcb863d921cb709e1b9a45baa0d9dd230
size 5898393792

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:9f9591941bbe8ccc65e76e9c71e1b3ed818648a1641bbb70672194c574e6c4bd
size 5569665216

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:64186a9d23e0503b2e7b8a988c3af738990fc87c8d89aaece29b60937ef0af1c
size 5435971776

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:87ac148fa3a88c3128705b4c19b64905461538e6d6f0cb47a2b14e3c1cee7006
size 6418618560

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:8529e618c1436f4ed8e75e46283c52fedc971899c006dc4ea8c46b572281379d
size 8312084672

245
README.md Normal file
View File

@@ -0,0 +1,245 @@
---
license: other
license_name: exaone
license_link: LICENSE
language:
- en
- ko
tags:
- lg-ai
- exaone
- exaone-3.5
pipeline_tag: text-generation
library_name: transformers
---
<br><img src="https://cdn-uploads.huggingface.co/production/uploads/646410e04bf9122922289dc7/b1ahV_SP1O43rTXACjmyz.webp" width="720"><br>
# Llama.cpp imatrix quantizations of [LGAI-EXAONE/EXAONE-3.5-7.8B-Instruct](https://huggingface.co/LGAI-EXAONE/EXAONE-3.5-7.8B-Instruct)
Using llama.cpp commit [5783575](https://github.com/ggerganov/llama.cpp/commit/5783575) for quantization.
All quants were made using the imatrix option and Bartowski's [calibration file](https://gist.github.com/bartowski1182/eb213dccb3571f863da82e99418f81e8).
<hr>
# Perplexity table (the lower the better)
| Quant | Size (MB) | PPL | Size (%) | Accuracy (%) | PPL error rate |
| ------------------------------------------------------------------------------------------------------------------------------ | --------- | ------- | -------- | ------------ | -------------- |
| [IQ1_S](https://huggingface.co/ThomasBaruzier/EXAONE-3.5-7.8B-Instruct-GGUF/blob/main/EXAONE-3.5-7.8B-Instruct-IQ1_S.gguf) | 1820 | 26.3205 | 12.20 | 33.81 | 0.40 |
| [IQ1_M](https://huggingface.co/ThomasBaruzier/EXAONE-3.5-7.8B-Instruct-GGUF/blob/main/EXAONE-3.5-7.8B-Instruct-IQ1_M.gguf) | 1955 | 19.0360 | 13.10 | 46.75 | 0.28 |
| [IQ2_XXS](https://huggingface.co/ThomasBaruzier/EXAONE-3.5-7.8B-Instruct-GGUF/blob/main/EXAONE-3.5-7.8B-Instruct-IQ2_XXS.gguf) | 2182 | 13.3276 | 14.63 | 66.77 | 0.20 |
| [IQ2_XS](https://huggingface.co/ThomasBaruzier/EXAONE-3.5-7.8B-Instruct-GGUF/blob/main/EXAONE-3.5-7.8B-Instruct-IQ2_XS.gguf) | 2379 | 11.7742 | 15.95 | 75.58 | 0.18 |
| [IQ2_S](https://huggingface.co/ThomasBaruzier/EXAONE-3.5-7.8B-Instruct-GGUF/blob/main/EXAONE-3.5-7.8B-Instruct-IQ2_S.gguf) | 2514 | 11.3084 | 16.85 | 78.69 | 0.17 |
| [IQ2_M](https://huggingface.co/ThomasBaruzier/EXAONE-3.5-7.8B-Instruct-GGUF/blob/main/EXAONE-3.5-7.8B-Instruct-IQ2_M.gguf) | 2695 | 10.3850 | 18.07 | 85.69 | 0.16 |
| [Q2_K_S](https://huggingface.co/ThomasBaruzier/EXAONE-3.5-7.8B-Instruct-GGUF/blob/main/EXAONE-3.5-7.8B-Instruct-Q2_K_S.gguf) | 2730 | 11.2910 | 18.30 | 78.82 | 0.17 |
| [Q2_K](https://huggingface.co/ThomasBaruzier/EXAONE-3.5-7.8B-Instruct-GGUF/blob/main/EXAONE-3.5-7.8B-Instruct-Q2_K.gguf) | 2912 | 11.1386 | 19.52 | 79.89 | 0.17 |
| [IQ3_XXS](https://huggingface.co/ThomasBaruzier/EXAONE-3.5-7.8B-Instruct-GGUF/blob/main/EXAONE-3.5-7.8B-Instruct-IQ3_XXS.gguf) | 3006 | 9.5453 | 20.15 | 93.23 | 0.14 |
| [IQ3_XS](https://huggingface.co/ThomasBaruzier/EXAONE-3.5-7.8B-Instruct-GGUF/blob/main/EXAONE-3.5-7.8B-Instruct-IQ3_XS.gguf) | 3226 | 9.2103 | 21.63 | 96.62 | 0.14 |
| [Q3_K_S](https://huggingface.co/ThomasBaruzier/EXAONE-3.5-7.8B-Instruct-GGUF/blob/main/EXAONE-3.5-7.8B-Instruct-Q3_K_S.gguf) | 3365 | 10.0571 | 22.56 | 88.49 | 0.16 |
| [IQ3_S](https://huggingface.co/ThomasBaruzier/EXAONE-3.5-7.8B-Instruct-GGUF/blob/main/EXAONE-3.5-7.8B-Instruct-IQ3_S.gguf) | 3382 | 9.2420 | 22.67 | 96.29 | 0.14 |
| [IQ3_M](https://huggingface.co/ThomasBaruzier/EXAONE-3.5-7.8B-Instruct-GGUF/blob/main/EXAONE-3.5-7.8B-Instruct-IQ3_M.gguf) | 3479 | 9.0709 | 23.32 | 98.11 | 0.13 |
| [Q3_K_M](https://huggingface.co/ThomasBaruzier/EXAONE-3.5-7.8B-Instruct-GGUF/blob/main/EXAONE-3.5-7.8B-Instruct-Q3_K_M.gguf) | 3703 | 9.2078 | 24.82 | 96.65 | 0.14 |
| [Q3_K_L](https://huggingface.co/ThomasBaruzier/EXAONE-3.5-7.8B-Instruct-GGUF/blob/main/EXAONE-3.5-7.8B-Instruct-Q3_K_L.gguf) | 3992 | 9.1908 | 26.76 | 96.83 | 0.14 |
| [IQ4_XS](https://huggingface.co/ThomasBaruzier/EXAONE-3.5-7.8B-Instruct-GGUF/blob/main/EXAONE-3.5-7.8B-Instruct-IQ4_XS.gguf) | 4101 | 9.0166 | 27.49 | 98.70 | 0.14 |
| [Q4_0](https://huggingface.co/ThomasBaruzier/EXAONE-3.5-7.8B-Instruct-GGUF/blob/main/EXAONE-3.5-7.8B-Instruct-Q4_0.gguf) | 4316 | 9.4186 | 28.93 | 94.49 | 0.14 |
| [IQ4_NL](https://huggingface.co/ThomasBaruzier/EXAONE-3.5-7.8B-Instruct-GGUF/blob/main/EXAONE-3.5-7.8B-Instruct-IQ4_NL.gguf) | 4318 | 9.0297 | 28.95 | 98.55 | 0.14 |
| [Q4_K_S](https://huggingface.co/ThomasBaruzier/EXAONE-3.5-7.8B-Instruct-GGUF/blob/main/EXAONE-3.5-7.8B-Instruct-Q4_K_S.gguf) | 4332 | 8.9634 | 29.04 | 99.28 | 0.13 |
| [Q4_K_M](https://huggingface.co/ThomasBaruzier/EXAONE-3.5-7.8B-Instruct-GGUF/blob/main/EXAONE-3.5-7.8B-Instruct-Q4_K_M.gguf) | 4549 | 8.9107 | 30.50 | 99.87 | 0.13 |
| [Q4_1](https://huggingface.co/ThomasBaruzier/EXAONE-3.5-7.8B-Instruct-GGUF/blob/main/EXAONE-3.5-7.8B-Instruct-Q4_1.gguf) | 4743 | 8.9614 | 31.80 | 99.31 | 0.13 |
| [Q5_K_S](https://huggingface.co/ThomasBaruzier/EXAONE-3.5-7.8B-Instruct-GGUF/blob/main/EXAONE-3.5-7.8B-Instruct-Q5_K_S.gguf) | 5184 | 8.9042 | 34.75 | 99.94 | 0.13 |
| [Q5_0](https://huggingface.co/ThomasBaruzier/EXAONE-3.5-7.8B-Instruct-GGUF/blob/main/EXAONE-3.5-7.8B-Instruct-Q5_0.gguf) | 5198 | 9.0533 | 34.85 | 98.30 | 0.14 |
| [Q5_K_M](https://huggingface.co/ThomasBaruzier/EXAONE-3.5-7.8B-Instruct-GGUF/blob/main/EXAONE-3.5-7.8B-Instruct-Q5_K_M.gguf) | 5311 | 8.9100 | 35.60 | 99.88 | 0.13 |
| [Q5_1](https://huggingface.co/ThomasBaruzier/EXAONE-3.5-7.8B-Instruct-GGUF/blob/main/EXAONE-3.5-7.8B-Instruct-Q5_1.gguf) | 5625 | 8.9230 | 37.71 | 99.73 | 0.13 |
| [Q6_K](https://huggingface.co/ThomasBaruzier/EXAONE-3.5-7.8B-Instruct-GGUF/blob/main/EXAONE-3.5-7.8B-Instruct-Q6_K.gguf) | 6121 | 8.8800 | 41.03 | 100.22 | 0.13 |
| [Q8_0](https://huggingface.co/ThomasBaruzier/EXAONE-3.5-7.8B-Instruct-GGUF/blob/main/EXAONE-3.5-7.8B-Instruct-Q8_0.gguf) | 7927 | 8.8534 | 53.14 | 100.52 | 0.13 |
| [F16](https://huggingface.co/ThomasBaruzier/EXAONE-3.5-7.8B-Instruct-GGUF/blob/main/EXAONE-3.5-7.8B-Instruct-F16.gguf) | 14917 | 8.8992 | 100 | 100 | 0.13 |
<hr>
<p align="center">
<img src="https://huggingface.co/LGAI-EXAONE/EXAONE-3.5-2.4B-Instruct/resolve/main/assets/EXAONE_Symbol+BI_3d.png", width="300", style="margin: 40 auto;">
<br>
# EXAONE-3.5-7.8B-Instruct
## Introduction
We introduce EXAONE 3.5, a collection of instruction-tuned bilingual (English and Korean) generative models ranging from 2.4B to 32B parameters, developed and released by LG AI Research. EXAONE 3.5 language models include: 1) **2.4B model** optimized for deployment on small or resource-constrained devices, 2) **7.8B model** matching the size of its predecessor but offering improved performance, and 3) **32B model** delivering powerful performance. All models support long-context processing of up to 32K tokens. Each model demonstrates state-of-the-art performance in real-world use cases and long-context understanding, while remaining competitive in general domains compared to recently released models of similar sizes.
For more details, please refer to our [technical report](https://arxiv.org/abs/2412.04862), [blog](https://www.lgresearch.ai/blog/view?seq=507) and [GitHub](https://github.com/LG-AI-EXAONE/EXAONE-3.5).
This repository contains the instruction-tuned 7.8B language model with the following features:
- Number of Parameters (without embeddings): 6.98B
- Number of Layers: 32
- Number of Attention Heads: GQA with 32 Q-heads and 8 KV-heads
- Vocab Size: 102,400
- Context Length: 32,768 tokens
## Quickstart
We recommend to use `transformers` v4.43 or later.
Here is the code snippet to run conversational inference with the model:
```python
import torch
from transformers import AutoModelForCausalLM, AutoTokenizer
model_name = "LGAI-EXAONE/EXAONE-3.5-7.8B-Instruct"
model = AutoModelForCausalLM.from_pretrained(
model_name,
torch_dtype=torch.bfloat16,
trust_remote_code=True,
device_map="auto"
)
tokenizer = AutoTokenizer.from_pretrained(model_name)
# Choose your prompt
prompt = "Explain how wonderful you are" # English example
prompt = "스스로를 자랑해 봐" # Korean example
messages = [
{"role": "system",
"content": "You are EXAONE model from LG AI Research, a helpful assistant."},
{"role": "user", "content": prompt}
]
input_ids = tokenizer.apply_chat_template(
messages,
tokenize=True,
add_generation_prompt=True,
return_tensors="pt"
)
output = model.generate(
input_ids.to("cuda"),
eos_token_id=tokenizer.eos_token_id,
max_new_tokens=128,
do_sample=False,
)
print(tokenizer.decode(output[0]))
```
> ### Note
> The EXAONE 3.5 instruction-tuned language models were trained to utilize the system prompt,
> so we highly recommend using the system prompts provided in the code snippet above.
## Evaluation
The following table shows the evaluation results of real-world use cases. The full evaluation results can be found in the [technical report](https://arxiv.org/abs/2412.04862).
<table>
<tr>
<th>Models</th>
<th>MT-Bench</th>
<th>LiveBench</th>
<th>Arena-Hard</th>
<th>AlpacaEval</th>
<th>IFEval</th>
<th>KoMT-Bench[1]</th>
<th>LogicKor</th>
</tr>
<tr>
<td>EXAONE 3.5 7.8B</td>
<td align="center"><strong>8.29</strong></td>
<td align="center"><strong>39.8</strong></td>
<td align="center"><strong>68.7</strong></td>
<td align="center"><strong>54.2</strong></td>
<td align="center"><strong>78.9</strong></td>
<td align="center"><strong>7.96</strong></td>
<td align="center"><strong>9.08</strong></td>
</tr>
<tr>
<td>Qwen 2.5 7B</td>
<td align="center">6.48</td>
<td align="center">35.6</td>
<td align="center">48.9</td>
<td align="center">31.7</td>
<td align="center">72.5</td>
<td align="center">5.19</td>
<td align="center">6.38</td>
</tr>
<tr>
<td>Llama 3.1 8B</td>
<td align="center">7.59</td>
<td align="center">28.3</td>
<td align="center">27.7</td>
<td align="center">25.7</td>
<td align="center">74.5</td>
<td align="center">4.85</td>
<td align="center">5.99</td>
</tr>
<tr>
<td>Gemma 2 9B</td>
<td align="center">7.64</td>
<td align="center">32.1</td>
<td align="center">43.6</td>
<td align="center">47.3</td>
<td align="center">54.7</td>
<td align="center">7.10</td>
<td align="center">8.05</td>
</tr>
<tr>
<td>Phi 3 small (7B)</td>
<td align="center">7.63</td>
<td align="center">27.9</td>
<td align="center">26.8</td>
<td align="center">29.2</td>
<td align="center">59.5</td>
<td align="center">3.22</td>
<td align="center">3.99</td>
</tr>
</table>
- [1] KoMT-Bench is a dataset created by translating MT-Bench into Korean; see [README](https://github.com/LG-AI-EXAONE/KoMT-Bench) for more details.
## Deployment
EXAONE 3.5 models can be inferred in the various frameworks, such as:
- `TensorRT-LLM`
- `vLLM`
- `SGLang`
- `llama.cpp`
- `Ollama`
Please refer to our [EXAONE 3.5 GitHub](https://github.com/LG-AI-EXAONE/EXAONE-3.5) for more details about the inference frameworks.
## Quantization
We provide the pre-quantized EXAONE 3.5 models with **AWQ** and several quantization types in **GGUF** format.
Please refer to our [EXAONE 3.5 collection](https://huggingface.co/collections/LGAI-EXAONE/exaone-35-674d0e1bb3dcd2ab6f39dbb4) to find corresponding quantized models.
## Limitation
The EXAONE language model has certain limitations and may occasionally generate inappropriate responses. The language model generates responses based on the output probability of tokens, and it is determined during learning from training data. While we have made every effort to exclude personal, harmful, and biased information from the training data, some problematic content may still be included, potentially leading to undesirable responses. Please note that the text generated by EXAONE language model does not reflects the views of LG AI Research.
- Inappropriate answers may be generated, which contain personal, harmful or other inappropriate information.
- Biased responses may be generated, which are associated with age, gender, race, and so on.
- The generated responses rely heavily on statistics from the training data, which can result in the generation of
semantically or syntactically incorrect sentences.
- Since the model does not reflect the latest information, the responses may be false or contradictory.
LG AI Research strives to reduce potential risks that may arise from EXAONE language models. Users are not allowed
to engage in any malicious activities (e.g., keying in illegal information) that may induce the creation of inappropriate
outputs violating LG AI’s ethical principles when using EXAONE language models.
## License
The model is licensed under [EXAONE AI Model License Agreement 1.1 - NC](./LICENSE)
## Citation
```
@article{exaone-3.5,
title={EXAONE 3.5: Series of Large Language Models for Real-world Use Cases},
author={LG AI Research},
journal={arXiv preprint arXiv:https://arxiv.org/abs/2412.04862},
year={2024}
}
```
## Contact
LG AI Research Technical Support: contact_us@lgresearch.ai

3
imatrix.dat Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:47a465a64d2bbf5e2581cd3b506a2902a95ee183f3e2eb6b3105e03a583b9be2
size 4988188

30
perplexity.md Normal file
View File

@@ -0,0 +1,30 @@
| Quant | Size (MB) | PPL | Size (%) | Accuracy (%) | PPL error rate |
| ------ | --------- | ------- | -------- | ------------ | -------------- |
| IQ1_S | 1820 | 26.3205 | 12.20 | 33.81 | 0.40 |
| IQ1_M | 1955 | 19.0360 | 13.10 | 46.75 | 0.28 |
| IQ2_XXS | 2182 | 13.3276 | 14.63 | 66.77 | 0.20 |
| IQ2_XS | 2379 | 11.7742 | 15.95 | 75.58 | 0.18 |
| IQ2_S | 2514 | 11.3084 | 16.85 | 78.69 | 0.17 |
| IQ2_M | 2695 | 10.3850 | 18.07 | 85.69 | 0.16 |
| Q2_K_S | 2730 | 11.2910 | 18.30 | 78.82 | 0.17 |
| Q2_K | 2912 | 11.1386 | 19.52 | 79.89 | 0.17 |
| IQ3_XXS | 3006 | 9.5453 | 20.15 | 93.23 | 0.14 |
| IQ3_XS | 3226 | 9.2103 | 21.63 | 96.62 | 0.14 |
| Q3_K_S | 3365 | 10.0571 | 22.56 | 88.49 | 0.16 |
| IQ3_S | 3382 | 9.2420 | 22.67 | 96.29 | 0.14 |
| IQ3_M | 3479 | 9.0709 | 23.32 | 98.11 | 0.13 |
| Q3_K_M | 3703 | 9.2078 | 24.82 | 96.65 | 0.14 |
| Q3_K_L | 3992 | 9.1908 | 26.76 | 96.83 | 0.14 |
| IQ4_XS | 4101 | 9.0166 | 27.49 | 98.70 | 0.14 |
| Q4_0 | 4316 | 9.4186 | 28.93 | 94.49 | 0.14 |
| IQ4_NL | 4318 | 9.0297 | 28.95 | 98.55 | 0.14 |
| Q4_K_S | 4332 | 8.9634 | 29.04 | 99.28 | 0.13 |
| Q4_K_M | 4549 | 8.9107 | 30.50 | 99.87 | 0.13 |
| Q4_1 | 4743 | 8.9614 | 31.80 | 99.31 | 0.13 |
| Q5_K_S | 5184 | 8.9042 | 34.75 | 99.94 | 0.13 |
| Q5_0 | 5198 | 9.0533 | 34.85 | 98.30 | 0.14 |
| Q5_K_M | 5311 | 8.9100 | 35.60 | 99.88 | 0.13 |
| Q5_1 | 5625 | 8.9230 | 37.71 | 99.73 | 0.13 |
| Q6_K | 6121 | 8.8800 | 41.03 | 100.22 | 0.13 |
| Q8_0 | 7927 | 8.8534 | 53.14 | 100.52 | 0.13 |
| F16 | 14917 | 8.8992 | 100 | 100 | 0.13 |