初始化项目,由ModelHub XC社区提供模型
Model: TheBloke/Sensei-7B-V1-AWQ Source: Original Platform
This commit is contained in:
35
.gitattributes
vendored
Normal file
35
.gitattributes
vendored
Normal file
@@ -0,0 +1,35 @@
|
|||||||
|
*.7z filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.arrow filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.bin filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.ftz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.gz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.h5 filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.joblib filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.model filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.npy filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.npz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.onnx filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.ot filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.parquet filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pb filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pickle filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pkl filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pt filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pth filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.rar filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||||
|
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tar filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tflite filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tgz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.wasm filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.xz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.zip filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.zst filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
||||||
94
LICENSE
Normal file
94
LICENSE
Normal file
@@ -0,0 +1,94 @@
|
|||||||
|
Business Source License 1.1
|
||||||
|
|
||||||
|
License text copyright (c) 2017 MariaDB Corporation Ab, All Rights Reserved.
|
||||||
|
"Business Source License" is a trademark of MariaDB Corporation Ab.
|
||||||
|
|
||||||
|
-----------------------------------------------------------------------------
|
||||||
|
|
||||||
|
Parameters
|
||||||
|
|
||||||
|
Licensor: Emergent AGI Inc.
|
||||||
|
|
||||||
|
Licensed Work: SciPhi/Sensei-7B-V1
|
||||||
|
The Licensed Work is (c) 2023 SciPhi
|
||||||
|
|
||||||
|
Change License: GNU General Public License v2.0 or later
|
||||||
|
|
||||||
|
-----------------------------------------------------------------------------
|
||||||
|
|
||||||
|
Terms
|
||||||
|
|
||||||
|
The Licensor hereby grants you the right to copy, modify, create derivative
|
||||||
|
works, redistribute, and make non-production use of the Licensed Work. The
|
||||||
|
Licensor may make an Additional Use Grant, above, permitting limited
|
||||||
|
production use.
|
||||||
|
|
||||||
|
Effective on the Change Date, or the fourth anniversary of the first publicly
|
||||||
|
available distribution of a specific version of the Licensed Work under this
|
||||||
|
License, whichever comes first, the Licensor hereby grants you rights under
|
||||||
|
the terms of the Change License, and the rights granted in the paragraph
|
||||||
|
above terminate.
|
||||||
|
|
||||||
|
If your use of the Licensed Work does not comply with the requirements
|
||||||
|
currently in effect as described in this License, you must purchase a
|
||||||
|
commercial license from the Licensor, its affiliated entities, or authorized
|
||||||
|
resellers, or you must refrain from using the Licensed Work.
|
||||||
|
|
||||||
|
All copies of the original and modified Licensed Work, and derivative works
|
||||||
|
of the Licensed Work, are subject to this License. This License applies
|
||||||
|
separately for each version of the Licensed Work and the Change Date may vary
|
||||||
|
for each version of the Licensed Work released by Licensor.
|
||||||
|
|
||||||
|
You must conspicuously display this License on each original or modified copy
|
||||||
|
of the Licensed Work. If you receive the Licensed Work in original or
|
||||||
|
modified form from a third party, the terms and conditions set forth in this
|
||||||
|
License apply to your use of that work.
|
||||||
|
|
||||||
|
Any use of the Licensed Work in violation of this License will automatically
|
||||||
|
terminate your rights under this License for the current and all other
|
||||||
|
versions of the Licensed Work.
|
||||||
|
|
||||||
|
This License does not grant you any right in any trademark or logo of
|
||||||
|
Licensor or its affiliates (provided that you may use a trademark or logo of
|
||||||
|
Licensor as expressly required by this License).
|
||||||
|
|
||||||
|
TO THE EXTENT PERMITTED BY APPLICABLE LAW, THE LICENSED WORK IS PROVIDED ON
|
||||||
|
AN "AS IS" BASIS. LICENSOR HEREBY DISCLAIMS ALL WARRANTIES AND CONDITIONS,
|
||||||
|
EXPRESS OR IMPLIED, INCLUDING (WITHOUT LIMITATION) WARRANTIES OF
|
||||||
|
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE, NON-INFRINGEMENT, AND
|
||||||
|
TITLE.
|
||||||
|
|
||||||
|
MariaDB hereby grants you permission to use this License’s text to license
|
||||||
|
your works, and to refer to it using the trademark "Business Source License",
|
||||||
|
as long as you comply with the Covenants of Licensor below.
|
||||||
|
|
||||||
|
-----------------------------------------------------------------------------
|
||||||
|
|
||||||
|
Covenants of Licensor
|
||||||
|
|
||||||
|
In consideration of the right to use this License’s text and the "Business
|
||||||
|
Source License" name and trademark, Licensor covenants to MariaDB, and to all
|
||||||
|
other recipients of the licensed work to be provided by Licensor:
|
||||||
|
|
||||||
|
1. To specify as the Change License the GPL Version 2.0 or any later version,
|
||||||
|
or a license that is compatible with GPL Version 2.0 or a later version,
|
||||||
|
where "compatible" means that software provided under the Change License can
|
||||||
|
be included in a program with software provided under GPL Version 2.0 or a
|
||||||
|
later version. Licensor may specify additional Change Licenses without
|
||||||
|
limitation.
|
||||||
|
|
||||||
|
2. To either: (a) specify an additional grant of rights to use that does not
|
||||||
|
impose any additional restriction on the right granted in this License, as
|
||||||
|
the Additional Use Grant; or (b) insert the text "None".
|
||||||
|
|
||||||
|
3. To specify a Change Date.
|
||||||
|
|
||||||
|
4. Not to modify this License in any other way.
|
||||||
|
|
||||||
|
-----------------------------------------------------------------------------
|
||||||
|
|
||||||
|
Notice
|
||||||
|
|
||||||
|
The Business Source License (this document, or the "License") is not an Open
|
||||||
|
Source license. However, the Licensed Work will eventually be made available
|
||||||
|
under an Open Source License, as stated in this License.
|
||||||
451
README.md
Normal file
451
README.md
Normal file
@@ -0,0 +1,451 @@
|
|||||||
|
---
|
||||||
|
base_model: SciPhi/Sensei-7B-V1
|
||||||
|
inference: false
|
||||||
|
model_creator: SciPhi-AI
|
||||||
|
model_name: Sensei 7B v1
|
||||||
|
model_type: mistral
|
||||||
|
prompt_template: "### Instruction: \nYour task is to perform retrieval augmented generation\
|
||||||
|
\ (RAG) over the given query and search results. Return your answer in a json format\
|
||||||
|
\ that includes a summary of the search results and a list of related queries. \n\
|
||||||
|
\nQuery:\n{prompt}\n\\n\\n\nSearch Results:\n{{context}}\n\\n\\n\nQuery:\n{prompt}\n\
|
||||||
|
\n### Response:\n{{\"summary\":\n"
|
||||||
|
quantized_by: TheBloke
|
||||||
|
---
|
||||||
|
<!-- markdownlint-disable MD041 -->
|
||||||
|
|
||||||
|
<!-- header start -->
|
||||||
|
<!-- 200823 -->
|
||||||
|
<div style="width: auto; margin-left: auto; margin-right: auto">
|
||||||
|
<img src="https://i.imgur.com/EBdldam.jpg" alt="TheBlokeAI" style="width: 100%; min-width: 400px; display: block; margin: auto;">
|
||||||
|
</div>
|
||||||
|
<div style="display: flex; justify-content: space-between; width: 100%;">
|
||||||
|
<div style="display: flex; flex-direction: column; align-items: flex-start;">
|
||||||
|
<p style="margin-top: 0.5em; margin-bottom: 0em;"><a href="https://discord.gg/theblokeai">Chat & support: TheBloke's Discord server</a></p>
|
||||||
|
</div>
|
||||||
|
<div style="display: flex; flex-direction: column; align-items: flex-end;">
|
||||||
|
<p style="margin-top: 0.5em; margin-bottom: 0em;"><a href="https://www.patreon.com/TheBlokeAI">Want to contribute? TheBloke's Patreon page</a></p>
|
||||||
|
</div>
|
||||||
|
</div>
|
||||||
|
<div style="text-align:center; margin-top: 0em; margin-bottom: 0em"><p style="margin-top: 0.25em; margin-bottom: 0em;">TheBloke's LLM work is generously supported by a grant from <a href="https://a16z.com">andreessen horowitz (a16z)</a></p></div>
|
||||||
|
<hr style="margin-top: 1.0em; margin-bottom: 1.0em;">
|
||||||
|
<!-- header end -->
|
||||||
|
|
||||||
|
# Sensei 7B v1 - AWQ
|
||||||
|
- Model creator: [SciPhi-AI](https://huggingface.co/SciPhi)
|
||||||
|
- Original model: [Sensei 7B v1](https://huggingface.co/SciPhi/Sensei-7B-V1)
|
||||||
|
|
||||||
|
<!-- description start -->
|
||||||
|
## Description
|
||||||
|
|
||||||
|
This repo contains AWQ model files for [SciPhi-AI's Sensei 7B v1](https://huggingface.co/SciPhi/Sensei-7B-V1).
|
||||||
|
|
||||||
|
These files were quantised using hardware kindly provided by [Massed Compute](https://massedcompute.com/).
|
||||||
|
|
||||||
|
|
||||||
|
### About AWQ
|
||||||
|
|
||||||
|
AWQ is an efficient, accurate and blazing-fast low-bit weight quantization method, currently supporting 4-bit quantization. Compared to GPTQ, it offers faster Transformers-based inference with equivalent or better quality compared to the most commonly used GPTQ settings.
|
||||||
|
|
||||||
|
AWQ models are currently supported on Linux and Windows, with NVidia GPUs only. macOS users: please use GGUF models instead.
|
||||||
|
|
||||||
|
It is supported by:
|
||||||
|
|
||||||
|
- [Text Generation Webui](https://github.com/oobabooga/text-generation-webui) - using Loader: AutoAWQ
|
||||||
|
- [vLLM](https://github.com/vllm-project/vllm) - version 0.2.2 or later for support for all model types.
|
||||||
|
- [Hugging Face Text Generation Inference (TGI)](https://github.com/huggingface/text-generation-inference)
|
||||||
|
- [Transformers](https://huggingface.co/docs/transformers) version 4.35.0 and later, from any code or client that supports Transformers
|
||||||
|
- [AutoAWQ](https://github.com/casper-hansen/AutoAWQ) - for use from Python code
|
||||||
|
|
||||||
|
<!-- description end -->
|
||||||
|
<!-- repositories-available start -->
|
||||||
|
## Repositories available
|
||||||
|
|
||||||
|
* [AWQ model(s) for GPU inference.](https://huggingface.co/TheBloke/Sensei-7B-V1-AWQ)
|
||||||
|
* [GPTQ models for GPU inference, with multiple quantisation parameter options.](https://huggingface.co/TheBloke/Sensei-7B-V1-GPTQ)
|
||||||
|
* [2, 3, 4, 5, 6 and 8-bit GGUF models for CPU+GPU inference](https://huggingface.co/TheBloke/Sensei-7B-V1-GGUF)
|
||||||
|
* [SciPhi-AI's original unquantised fp16 model in pytorch format, for GPU inference and for further conversions](https://huggingface.co/SciPhi/Sensei-7B-V1)
|
||||||
|
<!-- repositories-available end -->
|
||||||
|
|
||||||
|
<!-- prompt-template start -->
|
||||||
|
## Prompt template: Sensei-RAG
|
||||||
|
|
||||||
|
```
|
||||||
|
### Instruction:
|
||||||
|
Your task is to perform retrieval augmented generation (RAG) over the given query and search results. Return your answer in a json format that includes a summary of the search results and a list of related queries.
|
||||||
|
|
||||||
|
Query:
|
||||||
|
{prompt}
|
||||||
|
\n\n
|
||||||
|
Search Results:
|
||||||
|
{{context}}
|
||||||
|
\n\n
|
||||||
|
Query:
|
||||||
|
{prompt}
|
||||||
|
|
||||||
|
### Response:
|
||||||
|
{{"summary":
|
||||||
|
|
||||||
|
```
|
||||||
|
|
||||||
|
<!-- prompt-template end -->
|
||||||
|
|
||||||
|
|
||||||
|
<!-- README_AWQ.md-provided-files start -->
|
||||||
|
## Provided files, and AWQ parameters
|
||||||
|
|
||||||
|
I currently release 128g GEMM models only. The addition of group_size 32 models, and GEMV kernel models, is being actively considered.
|
||||||
|
|
||||||
|
Models are released as sharded safetensors files.
|
||||||
|
|
||||||
|
| Branch | Bits | GS | AWQ Dataset | Seq Len | Size |
|
||||||
|
| ------ | ---- | -- | ----------- | ------- | ---- |
|
||||||
|
| [main](https://huggingface.co/TheBloke/Sensei-7B-V1-AWQ/tree/main) | 4 | 128 | [VMware Open Instruct](https://huggingface.co/datasets/VMware/open-instruct/viewer/) | 4096 | 4.15 GB
|
||||||
|
|
||||||
|
<!-- README_AWQ.md-provided-files end -->
|
||||||
|
|
||||||
|
<!-- README_AWQ.md-text-generation-webui start -->
|
||||||
|
## How to easily download and use this model in [text-generation-webui](https://github.com/oobabooga/text-generation-webui)
|
||||||
|
|
||||||
|
Please make sure you're using the latest version of [text-generation-webui](https://github.com/oobabooga/text-generation-webui).
|
||||||
|
|
||||||
|
It is strongly recommended to use the text-generation-webui one-click-installers unless you're sure you know how to make a manual install.
|
||||||
|
|
||||||
|
1. Click the **Model tab**.
|
||||||
|
2. Under **Download custom model or LoRA**, enter `TheBloke/Sensei-7B-V1-AWQ`.
|
||||||
|
3. Click **Download**.
|
||||||
|
4. The model will start downloading. Once it's finished it will say "Done".
|
||||||
|
5. In the top left, click the refresh icon next to **Model**.
|
||||||
|
6. In the **Model** dropdown, choose the model you just downloaded: `Sensei-7B-V1-AWQ`
|
||||||
|
7. Select **Loader: AutoAWQ**.
|
||||||
|
8. Click Load, and the model will load and is now ready for use.
|
||||||
|
9. If you want any custom settings, set them and then click **Save settings for this model** followed by **Reload the Model** in the top right.
|
||||||
|
10. Once you're ready, click the **Text Generation** tab and enter a prompt to get started!
|
||||||
|
<!-- README_AWQ.md-text-generation-webui end -->
|
||||||
|
|
||||||
|
<!-- README_AWQ.md-use-from-vllm start -->
|
||||||
|
## Multi-user inference server: vLLM
|
||||||
|
|
||||||
|
Documentation on installing and using vLLM [can be found here](https://vllm.readthedocs.io/en/latest/).
|
||||||
|
|
||||||
|
- Please ensure you are using vLLM version 0.2 or later.
|
||||||
|
- When using vLLM as a server, pass the `--quantization awq` parameter.
|
||||||
|
|
||||||
|
For example:
|
||||||
|
|
||||||
|
```shell
|
||||||
|
python3 -m vllm.entrypoints.api_server --model TheBloke/Sensei-7B-V1-AWQ --quantization awq --dtype auto
|
||||||
|
```
|
||||||
|
|
||||||
|
- When using vLLM from Python code, again set `quantization=awq`.
|
||||||
|
|
||||||
|
For example:
|
||||||
|
|
||||||
|
```python
|
||||||
|
from vllm import LLM, SamplingParams
|
||||||
|
|
||||||
|
prompts = [
|
||||||
|
"Tell me about AI",
|
||||||
|
"Write a story about llamas",
|
||||||
|
"What is 291 - 150?",
|
||||||
|
"How much wood would a woodchuck chuck if a woodchuck could chuck wood?",
|
||||||
|
]
|
||||||
|
prompt_template=f'''### Instruction:
|
||||||
|
Your task is to perform retrieval augmented generation (RAG) over the given query and search results. Return your answer in a json format that includes a summary of the search results and a list of related queries.
|
||||||
|
|
||||||
|
Query:
|
||||||
|
{prompt}
|
||||||
|
\n\n
|
||||||
|
Search Results:
|
||||||
|
{{context}}
|
||||||
|
\n\n
|
||||||
|
Query:
|
||||||
|
{prompt}
|
||||||
|
|
||||||
|
### Response:
|
||||||
|
{{"summary":
|
||||||
|
'''
|
||||||
|
|
||||||
|
prompts = [prompt_template.format(prompt=prompt) for prompt in prompts]
|
||||||
|
|
||||||
|
sampling_params = SamplingParams(temperature=0.8, top_p=0.95)
|
||||||
|
|
||||||
|
llm = LLM(model="TheBloke/Sensei-7B-V1-AWQ", quantization="awq", dtype="auto")
|
||||||
|
|
||||||
|
outputs = llm.generate(prompts, sampling_params)
|
||||||
|
|
||||||
|
# Print the outputs.
|
||||||
|
for output in outputs:
|
||||||
|
prompt = output.prompt
|
||||||
|
generated_text = output.outputs[0].text
|
||||||
|
print(f"Prompt: {prompt!r}, Generated text: {generated_text!r}")
|
||||||
|
```
|
||||||
|
<!-- README_AWQ.md-use-from-vllm start -->
|
||||||
|
|
||||||
|
<!-- README_AWQ.md-use-from-tgi start -->
|
||||||
|
## Multi-user inference server: Hugging Face Text Generation Inference (TGI)
|
||||||
|
|
||||||
|
Use TGI version 1.1.0 or later. The official Docker container is: `ghcr.io/huggingface/text-generation-inference:1.1.0`
|
||||||
|
|
||||||
|
Example Docker parameters:
|
||||||
|
|
||||||
|
```shell
|
||||||
|
--model-id TheBloke/Sensei-7B-V1-AWQ --port 3000 --quantize awq --max-input-length 3696 --max-total-tokens 4096 --max-batch-prefill-tokens 4096
|
||||||
|
```
|
||||||
|
|
||||||
|
Example Python code for interfacing with TGI (requires [huggingface-hub](https://github.com/huggingface/huggingface_hub) 0.17.0 or later):
|
||||||
|
|
||||||
|
```shell
|
||||||
|
pip3 install huggingface-hub
|
||||||
|
```
|
||||||
|
|
||||||
|
```python
|
||||||
|
from huggingface_hub import InferenceClient
|
||||||
|
|
||||||
|
endpoint_url = "https://your-endpoint-url-here"
|
||||||
|
|
||||||
|
prompt = "Tell me about AI"
|
||||||
|
prompt_template=f'''### Instruction:
|
||||||
|
Your task is to perform retrieval augmented generation (RAG) over the given query and search results. Return your answer in a json format that includes a summary of the search results and a list of related queries.
|
||||||
|
|
||||||
|
Query:
|
||||||
|
{prompt}
|
||||||
|
\n\n
|
||||||
|
Search Results:
|
||||||
|
{{context}}
|
||||||
|
\n\n
|
||||||
|
Query:
|
||||||
|
{prompt}
|
||||||
|
|
||||||
|
### Response:
|
||||||
|
{{"summary":
|
||||||
|
'''
|
||||||
|
|
||||||
|
client = InferenceClient(endpoint_url)
|
||||||
|
response = client.text_generation(prompt,
|
||||||
|
max_new_tokens=128,
|
||||||
|
do_sample=True,
|
||||||
|
temperature=0.7,
|
||||||
|
top_p=0.95,
|
||||||
|
top_k=40,
|
||||||
|
repetition_penalty=1.1)
|
||||||
|
|
||||||
|
print(f"Model output: ", response)
|
||||||
|
```
|
||||||
|
<!-- README_AWQ.md-use-from-tgi end -->
|
||||||
|
|
||||||
|
<!-- README_AWQ.md-use-from-python start -->
|
||||||
|
## Inference from Python code using Transformers
|
||||||
|
|
||||||
|
### Install the necessary packages
|
||||||
|
|
||||||
|
- Requires: [Transformers](https://huggingface.co/docs/transformers) 4.35.0 or later.
|
||||||
|
- Requires: [AutoAWQ](https://github.com/casper-hansen/AutoAWQ) 0.1.6 or later.
|
||||||
|
|
||||||
|
```shell
|
||||||
|
pip3 install --upgrade "autoawq>=0.1.6" "transformers>=4.35.0"
|
||||||
|
```
|
||||||
|
|
||||||
|
Note that if you are using PyTorch 2.0.1, the above AutoAWQ command will automatically upgrade you to PyTorch 2.1.0.
|
||||||
|
|
||||||
|
If you are using CUDA 11.8 and wish to continue using PyTorch 2.0.1, instead run this command:
|
||||||
|
|
||||||
|
```shell
|
||||||
|
pip3 install https://github.com/casper-hansen/AutoAWQ/releases/download/v0.1.6/autoawq-0.1.6+cu118-cp310-cp310-linux_x86_64.whl
|
||||||
|
```
|
||||||
|
|
||||||
|
If you have problems installing [AutoAWQ](https://github.com/casper-hansen/AutoAWQ) using the pre-built wheels, install it from source instead:
|
||||||
|
|
||||||
|
```shell
|
||||||
|
pip3 uninstall -y autoawq
|
||||||
|
git clone https://github.com/casper-hansen/AutoAWQ
|
||||||
|
cd AutoAWQ
|
||||||
|
pip3 install .
|
||||||
|
```
|
||||||
|
|
||||||
|
### Transformers example code (requires Transformers 4.35.0 and later)
|
||||||
|
|
||||||
|
```python
|
||||||
|
from transformers import AutoModelForCausalLM, AutoTokenizer, TextStreamer
|
||||||
|
|
||||||
|
model_name_or_path = "TheBloke/Sensei-7B-V1-AWQ"
|
||||||
|
|
||||||
|
tokenizer = AutoTokenizer.from_pretrained(model_name_or_path)
|
||||||
|
model = AutoModelForCausalLM.from_pretrained(
|
||||||
|
model_name_or_path,
|
||||||
|
low_cpu_mem_usage=True,
|
||||||
|
device_map="cuda:0"
|
||||||
|
)
|
||||||
|
|
||||||
|
# Using the text streamer to stream output one token at a time
|
||||||
|
streamer = TextStreamer(tokenizer, skip_prompt=True, skip_special_tokens=True)
|
||||||
|
|
||||||
|
prompt = "Tell me about AI"
|
||||||
|
prompt_template=f'''### Instruction:
|
||||||
|
Your task is to perform retrieval augmented generation (RAG) over the given query and search results. Return your answer in a json format that includes a summary of the search results and a list of related queries.
|
||||||
|
|
||||||
|
Query:
|
||||||
|
{prompt}
|
||||||
|
\n\n
|
||||||
|
Search Results:
|
||||||
|
{{context}}
|
||||||
|
\n\n
|
||||||
|
Query:
|
||||||
|
{prompt}
|
||||||
|
|
||||||
|
### Response:
|
||||||
|
{{"summary":
|
||||||
|
'''
|
||||||
|
|
||||||
|
# Convert prompt to tokens
|
||||||
|
tokens = tokenizer(
|
||||||
|
prompt_template,
|
||||||
|
return_tensors='pt'
|
||||||
|
).input_ids.cuda()
|
||||||
|
|
||||||
|
generation_params = {
|
||||||
|
"do_sample": True,
|
||||||
|
"temperature": 0.7,
|
||||||
|
"top_p": 0.95,
|
||||||
|
"top_k": 40,
|
||||||
|
"max_new_tokens": 512,
|
||||||
|
"repetition_penalty": 1.1
|
||||||
|
}
|
||||||
|
|
||||||
|
# Generate streamed output, visible one token at a time
|
||||||
|
generation_output = model.generate(
|
||||||
|
tokens,
|
||||||
|
streamer=streamer,
|
||||||
|
**generation_params
|
||||||
|
)
|
||||||
|
|
||||||
|
# Generation without a streamer, which will include the prompt in the output
|
||||||
|
generation_output = model.generate(
|
||||||
|
tokens,
|
||||||
|
**generation_params
|
||||||
|
)
|
||||||
|
|
||||||
|
# Get the tokens from the output, decode them, print them
|
||||||
|
token_output = generation_output[0]
|
||||||
|
text_output = tokenizer.decode(token_output)
|
||||||
|
print("model.generate output: ", text_output)
|
||||||
|
|
||||||
|
# Inference is also possible via Transformers' pipeline
|
||||||
|
from transformers import pipeline
|
||||||
|
|
||||||
|
pipe = pipeline(
|
||||||
|
"text-generation",
|
||||||
|
model=model,
|
||||||
|
tokenizer=tokenizer,
|
||||||
|
**generation_params
|
||||||
|
)
|
||||||
|
|
||||||
|
pipe_output = pipe(prompt_template)[0]['generated_text']
|
||||||
|
print("pipeline output: ", pipe_output)
|
||||||
|
|
||||||
|
```
|
||||||
|
<!-- README_AWQ.md-use-from-python end -->
|
||||||
|
|
||||||
|
<!-- README_AWQ.md-compatibility start -->
|
||||||
|
## Compatibility
|
||||||
|
|
||||||
|
The files provided are tested to work with:
|
||||||
|
|
||||||
|
- [text-generation-webui](https://github.com/oobabooga/text-generation-webui) using `Loader: AutoAWQ`.
|
||||||
|
- [vLLM](https://github.com/vllm-project/vllm) version 0.2.0 and later.
|
||||||
|
- [Hugging Face Text Generation Inference (TGI)](https://github.com/huggingface/text-generation-inference) version 1.1.0 and later.
|
||||||
|
- [Transformers](https://huggingface.co/docs/transformers) version 4.35.0 and later.
|
||||||
|
- [AutoAWQ](https://github.com/casper-hansen/AutoAWQ) version 0.1.1 and later.
|
||||||
|
|
||||||
|
<!-- README_AWQ.md-compatibility end -->
|
||||||
|
|
||||||
|
<!-- footer start -->
|
||||||
|
<!-- 200823 -->
|
||||||
|
## Discord
|
||||||
|
|
||||||
|
For further support, and discussions on these models and AI in general, join us at:
|
||||||
|
|
||||||
|
[TheBloke AI's Discord server](https://discord.gg/theblokeai)
|
||||||
|
|
||||||
|
## Thanks, and how to contribute
|
||||||
|
|
||||||
|
Thanks to the [chirper.ai](https://chirper.ai) team!
|
||||||
|
|
||||||
|
Thanks to Clay from [gpus.llm-utils.org](llm-utils)!
|
||||||
|
|
||||||
|
I've had a lot of people ask if they can contribute. I enjoy providing models and helping people, and would love to be able to spend even more time doing it, as well as expanding into new projects like fine tuning/training.
|
||||||
|
|
||||||
|
If you're able and willing to contribute it will be most gratefully received and will help me to keep providing more models, and to start work on new AI projects.
|
||||||
|
|
||||||
|
Donaters will get priority support on any and all AI/LLM/model questions and requests, access to a private Discord room, plus other benefits.
|
||||||
|
|
||||||
|
* Patreon: https://patreon.com/TheBlokeAI
|
||||||
|
* Ko-Fi: https://ko-fi.com/TheBlokeAI
|
||||||
|
|
||||||
|
**Special thanks to**: Aemon Algiz.
|
||||||
|
|
||||||
|
**Patreon special mentions**: Michael Levine, 阿明, Trailburnt, Nikolai Manek, John Detwiler, Randy H, Will Dee, Sebastain Graf, NimbleBox.ai, Eugene Pentland, Emad Mostaque, Ai Maven, Jim Angel, Jeff Scroggin, Michael Davis, Manuel Alberto Morcote, Stephen Murray, Robert, Justin Joy, Luke @flexchar, Brandon Frisco, Elijah Stavena, S_X, Dan Guido, Undi ., Komninos Chatzipapas, Shadi, theTransient, Lone Striker, Raven Klaugh, jjj, Cap'n Zoog, Michel-Marie MAUDET (LINAGORA), Matthew Berman, David, Fen Risland, Omer Bin Jawed, Luke Pendergrass, Kalila, OG, Erik Bjäreholt, Rooh Singh, Joseph William Delisle, Dan Lewis, TL, John Villwock, AzureBlack, Brad, Pedro Madruga, Caitlyn Gatomon, K, jinyuan sun, Mano Prime, Alex, Jeffrey Morgan, Alicia Loh, Illia Dulskyi, Chadd, transmissions 11, fincy, Rainer Wilmers, ReadyPlayerEmma, knownsqashed, Mandus, biorpg, Deo Leter, Brandon Phillips, SuperWojo, Sean Connelly, Iucharbius, Jack West, Harry Royden McLaughlin, Nicholas, terasurfer, Vitor Caleffi, Duane Dunston, Johann-Peter Hartmann, David Ziegler, Olakabola, Ken Nordquist, Trenton Dambrowitz, Tom X Nguyen, Vadim, Ajan Kanaga, Leonard Tan, Clay Pascal, Alexandros Triantafyllidis, JM33133, Xule, vamX, ya boyyy, subjectnull, Talal Aujan, Alps Aficionado, wassieverse, Ari Malik, James Bentley, Woland, Spencer Kim, Michael Dempsey, Fred von Graf, Elle, zynix, William Richards, Stanislav Ovsiannikov, Edmond Seymore, Jonathan Leane, Martin Kemka, usrbinkat, Enrico Ros
|
||||||
|
|
||||||
|
|
||||||
|
Thank you to all my generous patrons and donaters!
|
||||||
|
|
||||||
|
And thank you again to a16z for their generous grant.
|
||||||
|
|
||||||
|
<!-- footer end -->
|
||||||
|
|
||||||
|
# Original model card: SciPhi-AI's Sensei 7B v1
|
||||||
|
|
||||||
|
|
||||||
|
# Sensei-7B-V1 Model Card
|
||||||
|
|
||||||
|
Sensei-7B-V1 is a Large Language Model (LLM) fine-tuned from OpenPipe's mistral-ft-optimized-1218, which is based on Mistral-7B. Sensei-7B-V1 was was fine-tuned with a fully synthetic dataset to specialize at performing retrieval-augmented generation (RAG) over detailed web search results. This model strives to specialize in using search, such as [AgentSearch](https://huggingface.co/datasets/SciPhi/AgentSearch-V1), to generate accurate and well-cited summaries from a range of search results, providing more accurate answers to user queries. Please refer to the [docs here](https://agent-search.readthedocs.io/en/latest/) for more information on how to run Sensei end-to-end.
|
||||||
|
|
||||||
|
Currently, Sensei is available via hosted api at https://www.sciphi.ai. You can try a demonstration [here](https://search.sciphi.ai/).
|
||||||
|
|
||||||
|
## Model Architecture
|
||||||
|
|
||||||
|
Base Model: mistral-ft-optimized-1218
|
||||||
|
|
||||||
|
**Architecture Features:**
|
||||||
|
- Transformer-based model
|
||||||
|
- Grouped-Query Attention
|
||||||
|
- Sliding-Window Attention
|
||||||
|
- Byte-fallback BPE tokenizer
|
||||||
|
|
||||||
|
|
||||||
|
## Using the Model
|
||||||
|
|
||||||
|
It is recommended to use a single search query. The model will return an answer using search results as context.
|
||||||
|
|
||||||
|
Using the AgentSearch package an example is shown below.
|
||||||
|
```
|
||||||
|
export SCIPHI_API_KEY=MY_SCIPHI_API_KEY
|
||||||
|
# Use `Sensei` for LLM RAG w/ AgentSearch
|
||||||
|
python -m agent_search.scripts.run_rag run --query="What is Fermat's last theorem?"
|
||||||
|
```
|
||||||
|
|
||||||
|
Alternatively, you may provide your own search context directly to the model by adhereing to the following format:
|
||||||
|
|
||||||
|
```
|
||||||
|
### Instruction:
|
||||||
|
Your task is to perform retrieval augmented generation (RAG) over the given query and search results. Return your answer in a json format that includes a summary of the search results and a list of related queries.
|
||||||
|
|
||||||
|
Query:
|
||||||
|
{prompt}
|
||||||
|
\n\n
|
||||||
|
Search Results:
|
||||||
|
{context}
|
||||||
|
\n\n
|
||||||
|
Query:
|
||||||
|
{prompt}
|
||||||
|
|
||||||
|
### Response:
|
||||||
|
{"summary":
|
||||||
|
```
|
||||||
|
|
||||||
|
__Note__: The inclusion of the text '{"summary":' following the Response footer is intentional. This ensures that the model responds with the proper json format, failure to include this leading prefix can cause small deviaitons. Combining the output with the leading string '{"summary":' results in a properly formatted JSON with keys 'summary' and 'other_queries'.
|
||||||
|
|
||||||
|
[<img src="https://raw.githubusercontent.com/OpenAccess-AI-Collective/axolotl/main/image/axolotl-badge-web.png" alt="Built with Axolotl" width="200" height="32"/>](https://github.com/OpenAccess-AI-Collective/axolotl)
|
||||||
|
|
||||||
|
## References
|
||||||
|
|
||||||
|
1. OpenPipe AI. (2023). Model Card for mistral-ft-optimized-1218. The mistral-ft-1218 Large Language Model (LLM) is a pretrained generative text model with 7 billion parameters optimized for downstream fine-tuning on a variety of tasks. For full details, please refer to the release blog post. Model Architecture: Transformer with Grouped-Query Attention, Sliding-Window Attention, and Byte-fallback BPE tokenizer. [Link](https://huggingface.co/OpenPipe/mistral-ft-optimized-1218)
|
||||||
35
config.json
Normal file
35
config.json
Normal file
@@ -0,0 +1,35 @@
|
|||||||
|
{
|
||||||
|
"_name_or_path": "/workspace/process/sciphi_sensei-7b-v1/source",
|
||||||
|
"architectures": [
|
||||||
|
"MistralForCausalLM"
|
||||||
|
],
|
||||||
|
"attention_dropout": 0.0,
|
||||||
|
"bos_token_id": 1,
|
||||||
|
"eos_token_id": 2,
|
||||||
|
"hidden_act": "silu",
|
||||||
|
"hidden_size": 4096,
|
||||||
|
"initializer_range": 0.02,
|
||||||
|
"intermediate_size": 14336,
|
||||||
|
"max_position_embeddings": 32768,
|
||||||
|
"model_type": "mistral",
|
||||||
|
"num_attention_heads": 32,
|
||||||
|
"num_hidden_layers": 32,
|
||||||
|
"num_key_value_heads": 8,
|
||||||
|
"pad_token_id": 0,
|
||||||
|
"pretraining_tp": 1,
|
||||||
|
"quantization_config": {
|
||||||
|
"bits": 4,
|
||||||
|
"group_size": 128,
|
||||||
|
"quant_method": "awq",
|
||||||
|
"version": "gemm",
|
||||||
|
"zero_point": true
|
||||||
|
},
|
||||||
|
"rms_norm_eps": 1e-05,
|
||||||
|
"rope_theta": 10000.0,
|
||||||
|
"sliding_window": 4096,
|
||||||
|
"tie_word_embeddings": false,
|
||||||
|
"torch_dtype": "float16",
|
||||||
|
"transformers_version": "4.35.2",
|
||||||
|
"use_cache": true,
|
||||||
|
"vocab_size": 32000
|
||||||
|
}
|
||||||
1
configuration.json
Normal file
1
configuration.json
Normal file
@@ -0,0 +1 @@
|
|||||||
|
{"framework": "pytorch", "task": "text-generation", "allow_remote": true}
|
||||||
6
generation_config.json
Normal file
6
generation_config.json
Normal file
@@ -0,0 +1,6 @@
|
|||||||
|
{
|
||||||
|
"_from_model_config": true,
|
||||||
|
"bos_token_id": 1,
|
||||||
|
"eos_token_id": 2,
|
||||||
|
"transformers_version": "4.36.2"
|
||||||
|
}
|
||||||
3
model.safetensors
Normal file
3
model.safetensors
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:1eab3190751cd650d46e410e603484678aca720628b05659f48f9de39485b70f
|
||||||
|
size 4150880232
|
||||||
6
quant_config.json
Normal file
6
quant_config.json
Normal file
@@ -0,0 +1,6 @@
|
|||||||
|
{
|
||||||
|
"zero_point": true,
|
||||||
|
"q_group_size": 128,
|
||||||
|
"w_bit": 4,
|
||||||
|
"version": "GEMM"
|
||||||
|
}
|
||||||
24
special_tokens_map.json
Normal file
24
special_tokens_map.json
Normal file
@@ -0,0 +1,24 @@
|
|||||||
|
{
|
||||||
|
"bos_token": {
|
||||||
|
"content": "<s>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false
|
||||||
|
},
|
||||||
|
"eos_token": {
|
||||||
|
"content": "</s>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false
|
||||||
|
},
|
||||||
|
"pad_token": "</s>",
|
||||||
|
"unk_token": {
|
||||||
|
"content": "<unk>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false
|
||||||
|
}
|
||||||
|
}
|
||||||
91122
tokenizer.json
Normal file
91122
tokenizer.json
Normal file
File diff suppressed because it is too large
Load Diff
BIN
tokenizer.model
(Stored with Git LFS)
Normal file
BIN
tokenizer.model
(Stored with Git LFS)
Normal file
Binary file not shown.
44
tokenizer_config.json
Normal file
44
tokenizer_config.json
Normal file
@@ -0,0 +1,44 @@
|
|||||||
|
{
|
||||||
|
"add_bos_token": true,
|
||||||
|
"add_eos_token": false,
|
||||||
|
"added_tokens_decoder": {
|
||||||
|
"0": {
|
||||||
|
"content": "<unk>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": true
|
||||||
|
},
|
||||||
|
"1": {
|
||||||
|
"content": "<s>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": true
|
||||||
|
},
|
||||||
|
"2": {
|
||||||
|
"content": "</s>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": true
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"additional_special_tokens": [],
|
||||||
|
"bos_token": "<s>",
|
||||||
|
"clean_up_tokenization_spaces": false,
|
||||||
|
"eos_token": "</s>",
|
||||||
|
"legacy": true,
|
||||||
|
"model_max_length": 1000000000000000019884624838656,
|
||||||
|
"pad_token": "</s>",
|
||||||
|
"sp_model_kwargs": {},
|
||||||
|
"spaces_between_special_tokens": false,
|
||||||
|
"tokenizer_class": "LlamaTokenizer",
|
||||||
|
"trust_remote_code": true,
|
||||||
|
"unk_token": "<unk>",
|
||||||
|
"use_default_system_prompt": false,
|
||||||
|
"use_fast": true
|
||||||
|
}
|
||||||
Reference in New Issue
Block a user