From 78cb5b31a22824794a57dfaa60220ba75cf8d3fd Mon Sep 17 00:00:00 2001 From: ModelHub XC Date: Mon, 27 Jul 2026 17:09:07 +0800 Subject: [PATCH] =?UTF-8?q?=E5=88=9D=E5=A7=8B=E5=8C=96=E9=A1=B9=E7=9B=AE?= =?UTF-8?q?=EF=BC=8C=E7=94=B1ModelHub=20XC=E7=A4=BE=E5=8C=BA=E6=8F=90?= =?UTF-8?q?=E4=BE=9B=E6=A8=A1=E5=9E=8B?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Model: Mungert/rwkv7-2.9B-world-GGUF Source: Original Platform --- .gitattributes | 76 +++++++++ README.md | 278 ++++++++++++++++++++++++++++++++ configuration.json | 1 + rwkv7-2.9B-world-bf16-q4_k.gguf | 3 + rwkv7-2.9B-world-bf16-q6_k.gguf | 3 + rwkv7-2.9B-world-bf16-q8_0.gguf | 3 + rwkv7-2.9B-world-bf16.gguf | 3 + rwkv7-2.9B-world-f16-q4_k.gguf | 3 + rwkv7-2.9B-world-f16-q6_k.gguf | 3 + rwkv7-2.9B-world-f16-q8_0.gguf | 3 + rwkv7-2.9B-world-iq3_m.gguf | 3 + rwkv7-2.9B-world-iq3_s.gguf | 3 + rwkv7-2.9B-world-iq3_xs.gguf | 3 + rwkv7-2.9B-world-iq3_xxs.gguf | 3 + rwkv7-2.9B-world-iq4_nl.gguf | 3 + rwkv7-2.9B-world-iq4_xs.gguf | 3 + rwkv7-2.9B-world-q3_k_m.gguf | 3 + rwkv7-2.9B-world-q3_k_s.gguf | 3 + rwkv7-2.9B-world-q4_0.gguf | 3 + rwkv7-2.9B-world-q4_1.gguf | 3 + rwkv7-2.9B-world-q4_k_m.gguf | 3 + rwkv7-2.9B-world-q4_k_s.gguf | 3 + rwkv7-2.9B-world-q5_0.gguf | 3 + rwkv7-2.9B-world-q5_1.gguf | 3 + rwkv7-2.9B-world-q5_k_m.gguf | 3 + rwkv7-2.9B-world-q5_k_s.gguf | 3 + rwkv7-2.9B-world-q6_k_m.gguf | 3 + rwkv7-2.9B-world-q8_0.gguf | 3 + rwkv7-2.9B-world-tq1_0.gguf | 3 + rwkv7-2.9B-world-tq2_0.gguf | 3 + rwkv7-2.9B-world.imatrix | 3 + 31 files changed, 439 insertions(+) create mode 100644 .gitattributes create mode 100644 README.md create mode 100644 configuration.json create mode 100644 rwkv7-2.9B-world-bf16-q4_k.gguf create mode 100644 rwkv7-2.9B-world-bf16-q6_k.gguf create mode 100644 rwkv7-2.9B-world-bf16-q8_0.gguf create mode 100644 rwkv7-2.9B-world-bf16.gguf create mode 100644 rwkv7-2.9B-world-f16-q4_k.gguf create mode 100644 rwkv7-2.9B-world-f16-q6_k.gguf create mode 100644 rwkv7-2.9B-world-f16-q8_0.gguf create mode 100644 rwkv7-2.9B-world-iq3_m.gguf create mode 100644 rwkv7-2.9B-world-iq3_s.gguf create mode 100644 rwkv7-2.9B-world-iq3_xs.gguf create mode 100644 rwkv7-2.9B-world-iq3_xxs.gguf create mode 100644 rwkv7-2.9B-world-iq4_nl.gguf create mode 100644 rwkv7-2.9B-world-iq4_xs.gguf create mode 100644 rwkv7-2.9B-world-q3_k_m.gguf create mode 100644 rwkv7-2.9B-world-q3_k_s.gguf create mode 100644 rwkv7-2.9B-world-q4_0.gguf create mode 100644 rwkv7-2.9B-world-q4_1.gguf create mode 100644 rwkv7-2.9B-world-q4_k_m.gguf create mode 100644 rwkv7-2.9B-world-q4_k_s.gguf create mode 100644 rwkv7-2.9B-world-q5_0.gguf create mode 100644 rwkv7-2.9B-world-q5_1.gguf create mode 100644 rwkv7-2.9B-world-q5_k_m.gguf create mode 100644 rwkv7-2.9B-world-q5_k_s.gguf create mode 100644 rwkv7-2.9B-world-q6_k_m.gguf create mode 100644 rwkv7-2.9B-world-q8_0.gguf create mode 100644 rwkv7-2.9B-world-tq1_0.gguf create mode 100644 rwkv7-2.9B-world-tq2_0.gguf create mode 100644 rwkv7-2.9B-world.imatrix diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..1399245 --- /dev/null +++ b/.gitattributes @@ -0,0 +1,76 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bin.* filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zstandard filter=lfs diff=lfs merge=lfs -text +*.tfevents* filter=lfs diff=lfs merge=lfs -text +*.db* filter=lfs diff=lfs merge=lfs -text +*.ark* filter=lfs diff=lfs merge=lfs -text +**/*ckpt*data* filter=lfs diff=lfs merge=lfs -text +**/*ckpt*.meta filter=lfs diff=lfs merge=lfs -text +**/*ckpt*.index filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text + +*.ggml filter=lfs diff=lfs merge=lfs -text +*.llamafile* filter=lfs diff=lfs merge=lfs -text +*.pt2 filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text + +rwkv7-2.9B-world-q5_1.gguf filter=lfs diff=lfs merge=lfs -text +rwkv7-2.9B-world-f16-q8_0.gguf filter=lfs diff=lfs merge=lfs -text +rwkv7-2.9B-world-q3_k_s.gguf filter=lfs diff=lfs merge=lfs -text +rwkv7-2.9B-world-iq4_nl.gguf filter=lfs diff=lfs merge=lfs -text +rwkv7-2.9B-world-bf16-q4_k.gguf filter=lfs diff=lfs merge=lfs -text +rwkv7-2.9B-world-iq3_xs.gguf filter=lfs diff=lfs merge=lfs -text +rwkv7-2.9B-world-f16-q4_k.gguf filter=lfs diff=lfs merge=lfs -text +rwkv7-2.9B-world-iq3_xxs.gguf filter=lfs diff=lfs merge=lfs -text +rwkv7-2.9B-world-iq3_s.gguf filter=lfs diff=lfs merge=lfs -text +rwkv7-2.9B-world-iq4_xs.gguf filter=lfs diff=lfs merge=lfs -text +rwkv7-2.9B-world-iq3_m.gguf filter=lfs diff=lfs merge=lfs -text +rwkv7-2.9B-world-q4_0.gguf filter=lfs diff=lfs merge=lfs -text +rwkv7-2.9B-world.imatrix filter=lfs diff=lfs merge=lfs -text +rwkv7-2.9B-world-bf16-q8_0.gguf filter=lfs diff=lfs merge=lfs -text +rwkv7-2.9B-world-q4_1.gguf filter=lfs diff=lfs merge=lfs -text +rwkv7-2.9B-world-tq2_0.gguf filter=lfs diff=lfs merge=lfs -text +rwkv7-2.9B-world-q4_k_s.gguf filter=lfs diff=lfs merge=lfs -text +rwkv7-2.9B-world-tq1_0.gguf filter=lfs diff=lfs merge=lfs -text +rwkv7-2.9B-world-q5_k_m.gguf filter=lfs diff=lfs merge=lfs -text +rwkv7-2.9B-world-q5_k_s.gguf filter=lfs diff=lfs merge=lfs -text +rwkv7-2.9B-world-q4_k_m.gguf filter=lfs diff=lfs merge=lfs -text +rwkv7-2.9B-world-q8_0.gguf filter=lfs diff=lfs merge=lfs -text +rwkv7-2.9B-world-q5_0.gguf filter=lfs diff=lfs merge=lfs -text +rwkv7-2.9B-world-bf16.gguf filter=lfs diff=lfs merge=lfs -text +rwkv7-2.9B-world-q6_k_m.gguf filter=lfs diff=lfs merge=lfs -text +rwkv7-2.9B-world-q3_k_m.gguf filter=lfs diff=lfs merge=lfs -text +rwkv7-2.9B-world-f16-q6_k.gguf filter=lfs diff=lfs merge=lfs -text +rwkv7-2.9B-world-bf16-q6_k.gguf filter=lfs diff=lfs merge=lfs -text \ No newline at end of file diff --git a/README.md b/README.md new file mode 100644 index 0000000..7c55ea4 --- /dev/null +++ b/README.md @@ -0,0 +1,278 @@ +--- +license: apache-2.0 +language: +- en +- zh +- ja +- ko +- fr +- ar +- es +- pt +metrics: +- accuracy +base_model: +- BlinkDL/rwkv-7-world +pipeline_tag: text-generation +--- + +# rwkv7-2.9B-world GGUF Models + +Note: you must use latest llama.cpp https://github.com/ggml-org/llama.cpp to run this model with llama.cp + +## **Choosing the Right Model Format** + +Selecting the correct model format depends on your **hardware capabilities** and **memory constraints**. + +### **BF16 (Brain Float 16) – Use if BF16 acceleration is available** +- A 16-bit floating-point format designed for **faster computation** while retaining good precision. +- Provides **similar dynamic range** as FP32 but with **lower memory usage**. +- Recommended if your hardware supports **BF16 acceleration** (check your device’s specs). +- Ideal for **high-performance inference** with **reduced memory footprint** compared to FP32. + +πŸ“Œ **Use BF16 if:** +βœ” Your hardware has native **BF16 support** (e.g., newer GPUs, TPUs). +βœ” You want **higher precision** while saving memory. +βœ” You plan to **requantize** the model into another format. + +πŸ“Œ **Avoid BF16 if:** +❌ Your hardware does **not** support BF16 (it may fall back to FP32 and run slower). +❌ You need compatibility with older devices that lack BF16 optimization. + +--- + +### **F16 (Float 16) – More widely supported than BF16** +- A 16-bit floating-point **high precision** but with less of range of values than BF16. +- Works on most devices with **FP16 acceleration support** (including many GPUs and some CPUs). +- Slightly lower numerical precision than BF16 but generally sufficient for inference. + +πŸ“Œ **Use F16 if:** +βœ” Your hardware supports **FP16** but **not BF16**. +βœ” You need a **balance between speed, memory usage, and accuracy**. +βœ” You are running on a **GPU** or another device optimized for FP16 computations. + +πŸ“Œ **Avoid F16 if:** +❌ Your device lacks **native FP16 support** (it may run slower than expected). +❌ You have memory limitations. + +--- + +### **Quantized Models (Q4_K, Q6_K, Q8, etc.) – For CPU & Low-VRAM Inference** +Quantization reduces model size and memory usage while maintaining as much accuracy as possible. +- **Lower-bit models (Q4_K)** β†’ **Best for minimal memory usage**, may have lower precision. +- **Higher-bit models (Q6_K, Q8_0)** β†’ **Better accuracy**, requires more memory. + +πŸ“Œ **Use Quantized Models if:** +βœ” You are running inference on a **CPU** and need an optimized model. +βœ” Your device has **low VRAM** and cannot load full-precision models. +βœ” You want to reduce **memory footprint** while keeping reasonable accuracy. + +πŸ“Œ **Avoid Quantized Models if:** +❌ You need **maximum accuracy** (full-precision models are better for this). +❌ Your hardware has enough VRAM for higher-precision formats (BF16/F16). + +--- + +### **Very Low-Bit Quantization (IQ3_XS, IQ3_S, IQ3_M, Q4_K, Q4_0)** +These models are optimized for **extreme memory efficiency**, making them ideal for **low-power devices** or **large-scale deployments** where memory is a critical constraint. + +- **IQ3_XS**: Ultra-low-bit quantization (3-bit) with **extreme memory efficiency**. + - **Use case**: Best for **ultra-low-memory devices** where even Q4_K is too large. + - **Trade-off**: Lower accuracy compared to higher-bit quantizations. + +- **IQ3_S**: Small block size for **maximum memory efficiency**. + - **Use case**: Best for **low-memory devices** where **IQ3_XS** is too aggressive. + +- **IQ3_M**: Medium block size for better accuracy than **IQ3_S**. + - **Use case**: Suitable for **low-memory devices** where **IQ3_S** is too limiting. + +- **Q4_K**: 4-bit quantization with **block-wise optimization** for better accuracy. + - **Use case**: Best for **low-memory devices** where **Q6_K** is too large. + +- **Q4_0**: Pure 4-bit quantization, optimized for **ARM devices**. + - **Use case**: Best for **ARM-based devices** or **low-memory environments**. + +--- + +### **Summary Table: Model Format Selection** + +| Model Format | Precision | Memory Usage | Device Requirements | Best Use Case | +|--------------|------------|---------------|----------------------|---------------| +| **BF16** | Highest | High | BF16-supported GPU/CPUs | High-speed inference with reduced memory | +| **F16** | High | High | FP16-supported devices | GPU inference when BF16 isn’t available | +| **Q4_K** | Medium Low | Low | CPU or Low-VRAM devices | Best for memory-constrained environments | +| **Q6_K** | Medium | Moderate | CPU with more memory | Better accuracy while still being quantized | +| **Q8_0** | High | Moderate | CPU or GPU with enough VRAM | Best accuracy among quantized models | +| **IQ3_XS** | Very Low | Very Low | Ultra-low-memory devices | Extreme memory efficiency and low accuracy | +| **Q4_0** | Low | Low | ARM or low-memory devices | llama.cpp can optimize for ARM devices | + +--- + +## **Included Files & Details** + +### `rwkv7-2.9B-world-bf16.gguf` +- Model weights preserved in **BF16**. +- Use this if you want to **requantize** the model into a different format. +- Best if your device supports **BF16 acceleration**. + +### `rwkv7-2.9B-world-f16.gguf` +- Model weights stored in **F16**. +- Use if your device supports **FP16**, especially if BF16 is not available. + +### `rwkv7-2.9B-world-bf16-q8_0.gguf` +- **Output & embeddings** remain in **BF16**. +- All other layers quantized to **Q8_0**. +- Use if your device supports **BF16** and you want a quantized version. + +### `rwkv7-2.9B-world-f16-q8_0.gguf` +- **Output & embeddings** remain in **F16**. +- All other layers quantized to **Q8_0**. + +### `rwkv7-2.9B-world-q4_k.gguf` +- **Output & embeddings** quantized to **Q8_0**. +- All other layers quantized to **Q4_K**. +- Good for **CPU inference** with limited memory. + +### `rwkv7-2.9B-world-q4_k_s.gguf` +- Smallest **Q4_K** variant, using less memory at the cost of accuracy. +- Best for **very low-memory setups**. + +### `rwkv7-2.9B-world-q6_k.gguf` +- **Output & embeddings** quantized to **Q8_0**. +- All other layers quantized to **Q6_K** . + +### `rwkv7-2.9B-world-q8_0.gguf` +- Fully **Q8** quantized model for better accuracy. +- Requires **more memory** but offers higher precision. + +### `rwkv7-2.9B-world-iq3_xs.gguf` +- **IQ3_XS** quantization, optimized for **extreme memory efficiency**. +- Best for **ultra-low-memory devices**. + +### `rwkv7-2.9B-world-iq3_m.gguf` +- **IQ3_M** quantization, offering a **medium block size** for better accuracy. +- Suitable for **low-memory devices**. + +### `rwkv7-2.9B-world-q4_0.gguf` +- Pure **Q4_0** quantization, optimized for **ARM devices**. +- Best for **low-memory environments**. +- Prefer IQ4_NL for better accuracy. + +# πŸš€ If you find these models useful + +Please click like ❀ . Also I’d really appreciate it if you could test my Network Monitor Assistant at πŸ‘‰ [Network Monitor Assitant](https://readyforquantum.com). + +πŸ’¬ Click the **chat icon** (bottom right of the main and dashboard pages) . Choose a LLM; toggle between the LLM Types TurboLLM -> FreeLLM -> TestLLM. + +### What I'm Testing + +I'm experimenting with **function calling** against my network monitoring service. Using small open source models. I am into the question "How small can it go and still function". + +🟑 **TestLLM** – Runs the current testing model using llama.cpp on 6 threads of a Cpu VM (Should take about 15s to load. Inference speed is quite slow and it only processes one user prompt at a timeβ€”still working on scaling!). If you're curious, I'd be happy to share how it works! . + +### The other Available AI Assistants + +🟒 **TurboLLM** – Uses **gpt-4o-mini** Fast! . Note: tokens are limited since OpenAI models are pricey, but you can [Login](https://readyforquantum.com) or [Download](https://readyforquantum.com/download/?utm_source=huggingface&utm_medium=referral&utm_campaign=huggingface_repo_readme) the Quantum Network Monitor agent to get more tokens, Alternatively use the TestLLM . + +πŸ”΅ **HugLLM** – Runs **open-source Hugging Face models** Fast, Runs small models (β‰ˆ8B) hence lower quality, Get 2x more tokens (subject to Hugging Face API availability) + +### Final Word + +I fund the servers used to create these model files, run the Quantum Network Monitor service, and pay for inference from Novita and OpenAIβ€”all out of my own pocket. All the code behind the model creation and the Quantum Network Monitor project is [open source](https://github.com/Mungert69). Feel free to use whatever you find helpful. + +If you appreciate the work, please consider [buying me a coffee](https://www.buymeacoffee.com/mahadeva) β˜•. Your support helps cover service costs and allows me to raise token limits for everyone. + +I'm also open to job opportunities or sponsorship. + +Thank you! 😊 + + + +# rwkv7-2.9B-world + + + +This is RWKV-7 model under flash-linear attention format. + +## Model Details + + +### Model Description + + + +- **Developed by:** Bo Peng, Yu Zhang, Songlin Yang, Ruichong Zhang +- **Funded by:** RWKV Project (Under LF AI & Data Foundation) +- **Model type:** RWKV7 +- **Language(s) (NLP):** English +- **License:** Apache-2.0 +- **Parameter count:** 2.9B +- **Tokenizer:** RWKV World tokenizer +- **Vocabulary size:** 65,536 + +### Model Sources + + + +- **Repository:** https://github.com/fla-org/flash-linear-attention ; https://github.com/BlinkDL/RWKV-LM +- **Paper:** With in Progress + +## Uses + + +Install `flash-linear-attention` and the latest version of `transformers` before using this model: + +```bash +pip install git+https://github.com/fla-org/flash-linear-attention +pip install 'transformers>=4.48.0' +``` + +### Direct Use + + +You can use this model just as any other HuggingFace models: +```python +from transformers import AutoModelForCausalLM, AutoTokenizer +model = AutoModelForCausalLM.from_pretrained('fla-hub/rwkv7-2.9B-world', trust_remote_code=True) +tokenizer = AutoTokenizer.from_pretrained('fla-hub/rwkv7-2.9B-world', trust_remote_code=True) +model = model.cuda() +prompt = "What is a large language model?" +messages = [ + {"role": "user", "content": "Who are you?"}, + {"role": "assistant", "content": "I am a GPT-3 based model."}, + {"role": "user", "content": prompt} +] +text = tokenizer.apply_chat_template( + messages, + tokenize=False, + add_generation_prompt=True +) + +model_inputs = tokenizer([text], return_tensors="pt").to(model.device) + +generated_ids = model.generate( + **model_inputs, + max_new_tokens=1024, +) +generated_ids = [ + output_ids[len(input_ids):] for input_ids, output_ids in zip(model_inputs.input_ids, generated_ids) +] + +response = tokenizer.batch_decode(generated_ids, skip_special_tokens=False)[0] +print(response) +``` + +### Training Data + +This model is trained on the World v3 with a total of 3.119 trillion tokens. + +#### Training Hyperparameters + +- **Training regime:** bfloat16, lr 4e-4 to 1e-5 "delayed" cosine decay, wd 0.1 (with increasing batch sizes during the middle) +- **Final Loss:** 1.8745 +- **Token Count:** 3.119 trillion + +## FAQ +Q: safetensors metadata is none. + +A: upgrade transformers to >=4.48.0: `pip install 'transformers>=4.48.0'` \ No newline at end of file diff --git a/configuration.json b/configuration.json new file mode 100644 index 0000000..bbeeda1 --- /dev/null +++ b/configuration.json @@ -0,0 +1 @@ +{"framework": "pytorch", "task": "text-generation", "allow_remote": true} \ No newline at end of file diff --git a/rwkv7-2.9B-world-bf16-q4_k.gguf b/rwkv7-2.9B-world-bf16-q4_k.gguf new file mode 100644 index 0000000..75cfed2 --- /dev/null +++ b/rwkv7-2.9B-world-bf16-q4_k.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d416a0bd83c0f649411994204f302c2f34594632f3f1778c2cb06669b7c64e25 +size 2314884640 diff --git a/rwkv7-2.9B-world-bf16-q6_k.gguf b/rwkv7-2.9B-world-bf16-q6_k.gguf new file mode 100644 index 0000000..1a83506 --- /dev/null +++ b/rwkv7-2.9B-world-bf16-q6_k.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:cd75e840cc0ba7f3c0d953a9dededb48733da751651b6d03758d3f2531193c95 +size 2963691040 diff --git a/rwkv7-2.9B-world-bf16-q8_0.gguf b/rwkv7-2.9B-world-bf16-q8_0.gguf new file mode 100644 index 0000000..3dba40c --- /dev/null +++ b/rwkv7-2.9B-world-bf16-q8_0.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:723333428c0093774c84d7952008aa4d919bf787517045cd38abb7035b072f80 +size 3573175584 diff --git a/rwkv7-2.9B-world-bf16.gguf b/rwkv7-2.9B-world-bf16.gguf new file mode 100644 index 0000000..0916266 --- /dev/null +++ b/rwkv7-2.9B-world-bf16.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:16246a65ffd7af86a84dac2185f9aa22c87085fb7c235174cd54f9d74da3e961 +size 5932471584 diff --git a/rwkv7-2.9B-world-f16-q4_k.gguf b/rwkv7-2.9B-world-f16-q4_k.gguf new file mode 100644 index 0000000..cfbdf1e --- /dev/null +++ b/rwkv7-2.9B-world-f16-q4_k.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f1f65fe174ca85aa3361b6d1fd6c8d3c251d9ee4f05bdb1e27b6a99a6230c99a +size 2314884640 diff --git a/rwkv7-2.9B-world-f16-q6_k.gguf b/rwkv7-2.9B-world-f16-q6_k.gguf new file mode 100644 index 0000000..4749025 --- /dev/null +++ b/rwkv7-2.9B-world-f16-q6_k.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1fa801f496a7aa6de0808f278b6251541b69b0ea563b709bac0c8072ac89fcd0 +size 2963691040 diff --git a/rwkv7-2.9B-world-f16-q8_0.gguf b/rwkv7-2.9B-world-f16-q8_0.gguf new file mode 100644 index 0000000..5f76096 --- /dev/null +++ b/rwkv7-2.9B-world-f16-q8_0.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d262267832d081d00de250dd9514b7b0496f3190f17d75d698532534bab91c29 +size 3573175584 diff --git a/rwkv7-2.9B-world-iq3_m.gguf b/rwkv7-2.9B-world-iq3_m.gguf new file mode 100644 index 0000000..489f984 --- /dev/null +++ b/rwkv7-2.9B-world-iq3_m.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:19db6c1ef7b8bb2633335fb787578e555966cd2d3d2985f4af600505186c8db9 +size 1519277600 diff --git a/rwkv7-2.9B-world-iq3_s.gguf b/rwkv7-2.9B-world-iq3_s.gguf new file mode 100644 index 0000000..f87e2db --- /dev/null +++ b/rwkv7-2.9B-world-iq3_s.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ddcc4b3566bcacb9d0c12aba34fbcdc6cbb4e2d0b0afa663b4db68cd34e6505c +size 1519277600 diff --git a/rwkv7-2.9B-world-iq3_xs.gguf b/rwkv7-2.9B-world-iq3_xs.gguf new file mode 100644 index 0000000..acf5140 --- /dev/null +++ b/rwkv7-2.9B-world-iq3_xs.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f91ef450518f21335674e542b22f8e06052bf352ae67d88c21e2e97a90ebf49b +size 1519277600 diff --git a/rwkv7-2.9B-world-iq3_xxs.gguf b/rwkv7-2.9B-world-iq3_xxs.gguf new file mode 100644 index 0000000..3e04d80 --- /dev/null +++ b/rwkv7-2.9B-world-iq3_xxs.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7fac008d19db659783b818c4b575cdb3884d433a5fcccaae7719dbb45dbe1ea5 +size 1379030560 diff --git a/rwkv7-2.9B-world-iq4_nl.gguf b/rwkv7-2.9B-world-iq4_nl.gguf new file mode 100644 index 0000000..734a300 --- /dev/null +++ b/rwkv7-2.9B-world-iq4_nl.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6026be63c2ec87ced6d569c53b07366ea9db3cebd2e8fe6c69283f98948d379c +size 1875793440 diff --git a/rwkv7-2.9B-world-iq4_xs.gguf b/rwkv7-2.9B-world-iq4_xs.gguf new file mode 100644 index 0000000..4eb7656 --- /dev/null +++ b/rwkv7-2.9B-world-iq4_xs.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b7b8599233c5c9b812a19126b5aee0cb71c451a6fe025914f14cc6bbeb4a9901 +size 1791907360 diff --git a/rwkv7-2.9B-world-q3_k_m.gguf b/rwkv7-2.9B-world-q3_k_m.gguf new file mode 100644 index 0000000..ceeb0c0 --- /dev/null +++ b/rwkv7-2.9B-world-q3_k_m.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7a9e549c8d6b236c6d566ae3561be89222c55c65103a2cd2039ec77572c2cbb7 +size 1519277600 diff --git a/rwkv7-2.9B-world-q3_k_s.gguf b/rwkv7-2.9B-world-q3_k_s.gguf new file mode 100644 index 0000000..10ab519 --- /dev/null +++ b/rwkv7-2.9B-world-q3_k_s.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7ae79ec04d6df1b844e12ab04adfc78948a3e4159f2c6bb24d753361fb95e47a +size 1519277600 diff --git a/rwkv7-2.9B-world-q4_0.gguf b/rwkv7-2.9B-world-q4_0.gguf new file mode 100644 index 0000000..5d21d78 --- /dev/null +++ b/rwkv7-2.9B-world-q4_0.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ff0a47f834cc5f0058be3f6d992a60721a0da3d02d0798398370ddf339341dfd +size 1832539680 diff --git a/rwkv7-2.9B-world-q4_1.gguf b/rwkv7-2.9B-world-q4_1.gguf new file mode 100644 index 0000000..d92d462 --- /dev/null +++ b/rwkv7-2.9B-world-q4_1.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e5d53f45885e185261eb17dea0d09e653b4e5daebf3dc5333ddd406ad1120fc8 +size 2010797600 diff --git a/rwkv7-2.9B-world-q4_k_m.gguf b/rwkv7-2.9B-world-q4_k_m.gguf new file mode 100644 index 0000000..2085aae --- /dev/null +++ b/rwkv7-2.9B-world-q4_k_m.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d80eadb5a5db0689a6e3a6689128670530b30203c44404510f77f553368253e8 +size 1875793440 diff --git a/rwkv7-2.9B-world-q4_k_s.gguf b/rwkv7-2.9B-world-q4_k_s.gguf new file mode 100644 index 0000000..2d63c0f --- /dev/null +++ b/rwkv7-2.9B-world-q4_k_s.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d2d403ad1d8da8c0303ebdea8bfd8bbf2ee62f2987881522ccdef7c167284419 +size 1875793440 diff --git a/rwkv7-2.9B-world-q5_0.gguf b/rwkv7-2.9B-world-q5_0.gguf new file mode 100644 index 0000000..2524d65 --- /dev/null +++ b/rwkv7-2.9B-world-q5_0.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:454d529d84a45f90c91914c3649ccaabf575102a437b1cfd41c6e819470cee92 +size 2189055520 diff --git a/rwkv7-2.9B-world-q5_1.gguf b/rwkv7-2.9B-world-q5_1.gguf new file mode 100644 index 0000000..0070aec --- /dev/null +++ b/rwkv7-2.9B-world-q5_1.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:87b7eaa642c198e395ceb7d66f6658f6e6944de883a813573290d76f996fccbd +size 2367313440 diff --git a/rwkv7-2.9B-world-q5_k_m.gguf b/rwkv7-2.9B-world-q5_k_m.gguf new file mode 100644 index 0000000..3e55c9b --- /dev/null +++ b/rwkv7-2.9B-world-q5_k_m.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ea260d859de6ec15e48d3cc35ecfc7d9f588060720cfad14951b83b17f420c6f +size 2211337760 diff --git a/rwkv7-2.9B-world-q5_k_s.gguf b/rwkv7-2.9B-world-q5_k_s.gguf new file mode 100644 index 0000000..33ea31f --- /dev/null +++ b/rwkv7-2.9B-world-q5_k_s.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ecc3386c0c8790e7d00a3cf22ef6c62327e8c0217364ee9a2a2b4a91e77c3faa +size 2211337760 diff --git a/rwkv7-2.9B-world-q6_k_m.gguf b/rwkv7-2.9B-world-q6_k_m.gguf new file mode 100644 index 0000000..16aaf72 --- /dev/null +++ b/rwkv7-2.9B-world-q6_k_m.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ff7156fae8ab92431545dc6f7a911bba1ebb386a2e9585dae74dcbeeded20d6e +size 2567853600 diff --git a/rwkv7-2.9B-world-q8_0.gguf b/rwkv7-2.9B-world-q8_0.gguf new file mode 100644 index 0000000..0870a42 --- /dev/null +++ b/rwkv7-2.9B-world-q8_0.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e52aeaffc7cfdfa38fdaf4fc8e864b8748ab522e123728bc6f406641aa2c1816 +size 3258602784 diff --git a/rwkv7-2.9B-world-tq1_0.gguf b/rwkv7-2.9B-world-tq1_0.gguf new file mode 100644 index 0000000..004f58e --- /dev/null +++ b/rwkv7-2.9B-world-tq1_0.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b3dca4d061adfe0f04d2b40feaf0c77ba1fd85da69c63edf194cee992073051a +size 991057440 diff --git a/rwkv7-2.9B-world-tq2_0.gguf b/rwkv7-2.9B-world-tq2_0.gguf new file mode 100644 index 0000000..10966ba --- /dev/null +++ b/rwkv7-2.9B-world-tq2_0.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:dd7f7fa980856302c726dd4472221b29c3b2e86f03fef9abdeec8019dc6ae1ca +size 1109022240 diff --git a/rwkv7-2.9B-world.imatrix b/rwkv7-2.9B-world.imatrix new file mode 100644 index 0000000..f00eaa1 --- /dev/null +++ b/rwkv7-2.9B-world.imatrix @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:49f46f0d3dbbd5e8e04d6ec7837ab6a0810e7bcd445554b2c837984811ea6b3f +size 4340297