commit b3b4e658850168b1f9a42efa4a8c669a633f0ab1 Author: ModelHub XC Date: Sun Jun 21 14:14:13 2026 +0800 初始化项目,由ModelHub XC社区提供模型 Model: QuantFactory/LLaMA-2-7B-32K-GGUF Source: Original Platform diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..df57330 --- /dev/null +++ b/.gitattributes @@ -0,0 +1,49 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text +LLaMA-2-7B-32K.Q4_K_M.gguf filter=lfs diff=lfs merge=lfs -text +LLaMA-2-7B-32K.Q4_1.gguf filter=lfs diff=lfs merge=lfs -text +LLaMA-2-7B-32K.Q4_0.gguf filter=lfs diff=lfs merge=lfs -text +LLaMA-2-7B-32K.Q8_0.gguf filter=lfs diff=lfs merge=lfs -text +LLaMA-2-7B-32K.Q6_K.gguf filter=lfs diff=lfs merge=lfs -text +LLaMA-2-7B-32K.Q5_K_M.gguf filter=lfs diff=lfs merge=lfs -text +LLaMA-2-7B-32K.Q5_K_S.gguf filter=lfs diff=lfs merge=lfs -text +LLaMA-2-7B-32K.Q5_0.gguf filter=lfs diff=lfs merge=lfs -text +LLaMA-2-7B-32K.Q5_1.gguf filter=lfs diff=lfs merge=lfs -text +LLaMA-2-7B-32K.Q2_K.gguf filter=lfs diff=lfs merge=lfs -text +LLaMA-2-7B-32K.Q3_K_L.gguf filter=lfs diff=lfs merge=lfs -text +LLaMA-2-7B-32K.Q3_K_M.gguf filter=lfs diff=lfs merge=lfs -text +LLaMA-2-7B-32K.Q3_K_S.gguf filter=lfs diff=lfs merge=lfs -text +LLaMA-2-7B-32K.Q4_K_S.gguf filter=lfs diff=lfs merge=lfs -text diff --git a/LLaMA-2-7B-32K.Q2_K.gguf b/LLaMA-2-7B-32K.Q2_K.gguf new file mode 100644 index 0000000..bf48750 --- /dev/null +++ b/LLaMA-2-7B-32K.Q2_K.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7156995f931a3a4fe12de46a6bfdb0d185d2944195dfefbcbdc4756a6d920bb6 +size 2532864352 diff --git a/LLaMA-2-7B-32K.Q3_K_L.gguf b/LLaMA-2-7B-32K.Q3_K_L.gguf new file mode 100644 index 0000000..7a0f073 --- /dev/null +++ b/LLaMA-2-7B-32K.Q3_K_L.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:83a7fe826f4f357e208864d9163fb85987ebade93b07529a9cbf325c5aa67157 +size 3597111648 diff --git a/LLaMA-2-7B-32K.Q3_K_M.gguf b/LLaMA-2-7B-32K.Q3_K_M.gguf new file mode 100644 index 0000000..3ec1cd6 --- /dev/null +++ b/LLaMA-2-7B-32K.Q3_K_M.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:19771be5de3192276c8da0c50448f8a0d383102d52a1c231e934f0ab0843e454 +size 3298005344 diff --git a/LLaMA-2-7B-32K.Q3_K_S.gguf b/LLaMA-2-7B-32K.Q3_K_S.gguf new file mode 100644 index 0000000..f151c83 --- /dev/null +++ b/LLaMA-2-7B-32K.Q3_K_S.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d4b8148447de20ca6a4abbbc01e13569f6733381e39a9d45a0d7cedcc3b6e219 +size 2948305248 diff --git a/LLaMA-2-7B-32K.Q4_0.gguf b/LLaMA-2-7B-32K.Q4_0.gguf new file mode 100644 index 0000000..a3e4265 --- /dev/null +++ b/LLaMA-2-7B-32K.Q4_0.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:cfa2256937f0890f10ae3beffa0d11fc30dc1ed8163456311aeb7c32f2dc2678 +size 3825807712 diff --git a/LLaMA-2-7B-32K.Q4_1.gguf b/LLaMA-2-7B-32K.Q4_1.gguf new file mode 100644 index 0000000..471d01d --- /dev/null +++ b/LLaMA-2-7B-32K.Q4_1.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a8f3981531379d501f64ad49eaf2ecffbc52e7cf44d64a1481b51f192710436f +size 4238750048 diff --git a/LLaMA-2-7B-32K.Q4_K_M.gguf b/LLaMA-2-7B-32K.Q4_K_M.gguf new file mode 100644 index 0000000..d2d0348 --- /dev/null +++ b/LLaMA-2-7B-32K.Q4_K_M.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:109843b0348f8c15b5ef16d3918ab42485e209cd65736e77fd7ad99b89597a78 +size 4081004896 diff --git a/LLaMA-2-7B-32K.Q4_K_S.gguf b/LLaMA-2-7B-32K.Q4_K_S.gguf new file mode 100644 index 0000000..7ef65af --- /dev/null +++ b/LLaMA-2-7B-32K.Q4_K_S.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6848ec2b61621f1c94c958191a21e2d130177aa7fa4cb7eb14084a5a683119b9 +size 3856740704 diff --git a/LLaMA-2-7B-32K.Q5_0.gguf b/LLaMA-2-7B-32K.Q5_0.gguf new file mode 100644 index 0000000..dbf8367 --- /dev/null +++ b/LLaMA-2-7B-32K.Q5_0.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:076f0fa2e4d4f2d7a39dbaad03a0b76d68a04593a304f2d362458ac6fbfbdb93 +size 4651692384 diff --git a/LLaMA-2-7B-32K.Q5_1.gguf b/LLaMA-2-7B-32K.Q5_1.gguf new file mode 100644 index 0000000..f8144a6 --- /dev/null +++ b/LLaMA-2-7B-32K.Q5_1.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:555a41382d08d15b61059c53645620fd334302d118f6bbb4f8ea998a9caef206 +size 5064634720 diff --git a/LLaMA-2-7B-32K.Q5_K_M.gguf b/LLaMA-2-7B-32K.Q5_K_M.gguf new file mode 100644 index 0000000..79b4a6b --- /dev/null +++ b/LLaMA-2-7B-32K.Q5_K_M.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:75dc800328fb536a9188c4e4a7d39617cfe1c73a564ed20199cf98948fb85bfb +size 4783157600 diff --git a/LLaMA-2-7B-32K.Q5_K_S.gguf b/LLaMA-2-7B-32K.Q5_K_S.gguf new file mode 100644 index 0000000..ca24324 --- /dev/null +++ b/LLaMA-2-7B-32K.Q5_K_S.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a06dce6aef1ad09cedaa0e9ed7241dc06b9b72704122ed29b63347efb131f3d9 +size 4651692384 diff --git a/LLaMA-2-7B-32K.Q6_K.gguf b/LLaMA-2-7B-32K.Q6_K.gguf new file mode 100644 index 0000000..d3533ff --- /dev/null +++ b/LLaMA-2-7B-32K.Q6_K.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:022833e37ea44271a89384c2824e139641f6555f50e6cf344633a745fc518289 +size 5529194848 diff --git a/LLaMA-2-7B-32K.Q8_0.gguf b/LLaMA-2-7B-32K.Q8_0.gguf new file mode 100644 index 0000000..a87dc51 --- /dev/null +++ b/LLaMA-2-7B-32K.Q8_0.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:54e3789e6558ec1ae90a79e3884f19e0a4d309feaf44180ee001a13015db6b5d +size 7161090400 diff --git a/README.md b/README.md new file mode 100644 index 0000000..dd8f90c --- /dev/null +++ b/README.md @@ -0,0 +1,121 @@ + +--- + +license: llama2 +datasets: +- togethercomputer/RedPajama-Data-1T +- togethercomputer/RedPajama-Data-Instruct +- EleutherAI/pile +- togethercomputer/Long-Data-Collections +language: +- en +library_name: transformers + +--- + +[![QuantFactory Banner](https://lh7-rt.googleusercontent.com/docsz/AD_4nXeiuCm7c8lEwEJuRey9kiVZsRn2W-b4pWlu3-X534V3YmVuVc2ZL-NXg2RkzSOOS2JXGHutDuyyNAUtdJI65jGTo8jT9Y99tMi4H4MqL44Uc5QKG77B0d6-JfIkZHFaUA71-RtjyYZWVIhqsNZcx8-OMaA?key=xt3VSDoCbmTY7o-cwwOFwQ)](https://hf.co/QuantFactory) + + +# QuantFactory/LLaMA-2-7B-32K-GGUF +This is quantized version of [togethercomputer/LLaMA-2-7B-32K](https://huggingface.co/togethercomputer/LLaMA-2-7B-32K) created using llama.cpp + +# Original Model Card + + +# LLaMA-2-7B-32K + +## Model Description + +LLaMA-2-7B-32K is an open-source, long context language model developed by Together, fine-tuned from Meta's original Llama-2 7B model. +This model represents our efforts to contribute to the rapid progress of the open-source ecosystem for large language models. +The model has been extended to a context length of 32K with position interpolation, +allowing applications on multi-document QA, long text summarization, etc. + +## What's new? + +This model introduces several improvements and new features: + +1. **Extended Context:** The model has been trained to handle context lengths up to 32K, which is a significant improvement over the previous versions. + +2. **Pre-training and Instruction Tuning:** We have shared our data recipe, which consists of a mixture of pre-training and instruction tuning data. + +3. **Fine-tuning Examples:** We provide examples of how to fine-tune the model for specific applications, including book summarization and long context question and answering. + +4. **Software Support:** We have updated both the inference and training stack to allow efficient inference and fine-tuning for 32K context. + +## Model Architecture + +The model follows the architecture of Llama-2-7B and extends it to handle a longer context. It leverages the recently released FlashAttention-2 and a range of other optimizations to improve the speed and efficiency of inference and training. + +## Training and Fine-tuning + +The model has been trained using a mixture of pre-training and instruction tuning data. +- In the first training phase of continued pre-training, our data mixture contains 25% RedPajama Book, 25% RedPajama ArXiv (including abstracts), 25% other data from RedPajama, and 25% from the UL2 Oscar Data, which is a part of OIG (Open-Instruction-Generalist), asking the model to fill in missing chunks, or complete the text. +To enhance the long-context ability, we exclude data shorter than 2K word. The inclusion of UL2 Oscar Data is effective in compelling the model to read and utilize long-range context. +- We then fine-tune the model to focus on its few shot capacity under long context, including 20% Natural Instructions (NI), 20% Public Pool of Prompts (P3), 20% the Pile. We decontaminated all data against HELM core scenarios . We teach the model to leverage the in-context examples by packing examples into one 32K-token sequence. To maintain the knowledge learned from the first piece of data, we incorporate 20% RedPajama-Data Book and 20% RedPajama-Data ArXiv. + +Next, we provide examples of how to fine-tune the model for specific applications. +The example datasets are placed in [togethercomputer/Long-Data-Collections](https://huggingface.co/datasets/togethercomputer/Long-Data-Collections) +You can use the [OpenChatKit](https://github.com/togethercomputer/OpenChatKit) to fine-tune your own 32K model over LLaMA-2-7B-32K. +Please refer to [OpenChatKit](https://github.com/togethercomputer/OpenChatKit) for step-by-step illustrations. + +1. Long Context QA. + + We take as an example the multi-document question answering task from the paper “Lost in the Middle: How Language Models Use Long Contexts”. The input for the model consists of (i) a question that requires an answer and (ii) k documents, which are passages extracted from Wikipedia. Notably, only one of these documents contains the answer to the question, while the remaining k − 1 documents, termed as "distractor" documents, do not. To successfully perform this task, the model must identify and utilize the document containing the answer from its input context. + + With OCK, simply run the following command to fine-tune: + ``` + bash training/finetune_llama-2-7b-32k-mqa.sh + ``` + +2. Summarization. + + Another example is BookSum, a unique dataset designed to address the challenges of long-form narrative summarization. This dataset features source documents from the literature domain, including novels, plays, and stories, and offers human-written, highly abstractive summaries. We here focus on chapter-level data. BookSum poses a unique set of challenges, necessitating that the model comprehensively read through each chapter. + + With OCK, simply run the following command to fine-tune: + ``` + bash training/finetune_llama-2-7b-32k-booksum.sh + ``` + + +## Inference + +You can use the [Together API](https://together.ai/blog/api-announcement) to try out LLaMA-2-7B-32K for inference. +The updated inference stack allows for efficient inference. + +To run the model locally, we strongly recommend to install Flash Attention V2, which is necessary to obtain the best performance: +``` +# Please update the path of `CUDA_HOME` +export CUDA_HOME=/usr/local/cuda-11.8 +pip install transformers==4.31.0 +pip install sentencepiece +pip install ninja +pip install flash-attn --no-build-isolation +pip install git+https://github.com/HazyResearch/flash-attention.git#subdirectory=csrc/rotary +``` + +You can use this model directly from the Hugging Face Model Hub or fine-tune it on your own data using the OpenChatKit. + +```python +from transformers import AutoTokenizer, AutoModelForCausalLM + +tokenizer = AutoTokenizer.from_pretrained("togethercomputer/LLaMA-2-7B-32K") +model = AutoModelForCausalLM.from_pretrained("togethercomputer/LLaMA-2-7B-32K", trust_remote_code=True, torch_dtype=torch.float16) + +input_context = "Your text here" +input_ids = tokenizer.encode(input_context, return_tensors="pt") +output = model.generate(input_ids, max_length=128, temperature=0.7) +output_text = tokenizer.decode(output[0], skip_special_tokens=True) +print(output_text) +``` + +Alternatively, you can set `trust_remote_code=False` if you prefer not to use flash attention. + + +## Limitations and Bias + +As with all language models, LLaMA-2-7B-32K may generate incorrect or biased content. It's important to keep this in mind when using the model. + +## Community + +Join us on [Together Discord](https://discord.gg/6ZVDU8tTD4) diff --git a/configuration.json b/configuration.json new file mode 100644 index 0000000..159097f --- /dev/null +++ b/configuration.json @@ -0,0 +1 @@ +{"framework": "pytorch", "task": "others", "allow_remote": true} \ No newline at end of file