commit 0c1a292fec1efc957159fe97bf7f12d813e6b663 Author: ModelHub XC Date: Fri Sep 4 15:21:41 2026 +0800 初始化项目,由ModelHub XC社区提供模型 Model: prithivMLmods/tiiuae_Falcon-H1R-7B-GGUF Source: Original Platform diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..fd09374 --- /dev/null +++ b/.gitattributes @@ -0,0 +1,61 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bin.* filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zstandard filter=lfs diff=lfs merge=lfs -text +*.tfevents* filter=lfs diff=lfs merge=lfs -text +*.db* filter=lfs diff=lfs merge=lfs -text +*.ark* filter=lfs diff=lfs merge=lfs -text +**/*ckpt*data* filter=lfs diff=lfs merge=lfs -text +**/*ckpt*.meta filter=lfs diff=lfs merge=lfs -text +**/*ckpt*.index filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text + +*.ggml filter=lfs diff=lfs merge=lfs -text +*.llamafile* filter=lfs diff=lfs merge=lfs -text +*.pt2 filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text + +Falcon-H1R-7B.Q5_K_S.gguf filter=lfs diff=lfs merge=lfs -text +Falcon-H1R-7B.Q3_K_S.gguf filter=lfs diff=lfs merge=lfs -text +Falcon-H1R-7B.Q3_K_L.gguf filter=lfs diff=lfs merge=lfs -text +Falcon-H1R-7B.Q2_K.gguf filter=lfs diff=lfs merge=lfs -text +Falcon-H1R-7B.Q8_0.gguf filter=lfs diff=lfs merge=lfs -text +Falcon-H1R-7B-bf16.gguf filter=lfs diff=lfs merge=lfs -text +Falcon-H1R-7B.Q3_K_M.gguf filter=lfs diff=lfs merge=lfs -text +Falcon-H1R-7B.IQ4_XS.gguf filter=lfs diff=lfs merge=lfs -text +Falcon-H1R-7B.f16.gguf filter=lfs diff=lfs merge=lfs -text +Falcon-H1R-7B.Q4_K_M.gguf filter=lfs diff=lfs merge=lfs -text +Falcon-H1R-7B-f32.gguf filter=lfs diff=lfs merge=lfs -text +Falcon-H1R-7B.Q6_K.gguf filter=lfs diff=lfs merge=lfs -text +Falcon-H1R-7B.Q4_K_S.gguf filter=lfs diff=lfs merge=lfs -text \ No newline at end of file diff --git a/Falcon-H1R-7B-bf16.gguf b/Falcon-H1R-7B-bf16.gguf new file mode 100644 index 0000000..95c5452 --- /dev/null +++ b/Falcon-H1R-7B-bf16.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ac852a921c550ec9c55490c2a7aabd6706843e86b1a8cb9c31d5da908245f77e +size 15179424128 diff --git a/Falcon-H1R-7B-f32.gguf b/Falcon-H1R-7B-f32.gguf new file mode 100644 index 0000000..1216093 --- /dev/null +++ b/Falcon-H1R-7B-f32.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8c7a48fec169a959e1a5ce60483669fb4b75088cf2441d931a059337d6f0d55f +size 30348321152 diff --git a/Falcon-H1R-7B.IQ4_XS.gguf b/Falcon-H1R-7B.IQ4_XS.gguf new file mode 100644 index 0000000..023d4fc --- /dev/null +++ b/Falcon-H1R-7B.IQ4_XS.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8b476b7839314536fcd780cf55ebc9ddc985ade14ecd2843048d40820b62e186 +size 4190146240 diff --git a/Falcon-H1R-7B.Q2_K.gguf b/Falcon-H1R-7B.Q2_K.gguf new file mode 100644 index 0000000..4e84c30 --- /dev/null +++ b/Falcon-H1R-7B.Q2_K.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d53f02799694d00cf54d7bc1698db324a4bf6e5c9429856cc5aaf7fe8c9ee87c +size 2893693120 diff --git a/Falcon-H1R-7B.Q3_K_L.gguf b/Falcon-H1R-7B.Q3_K_L.gguf new file mode 100644 index 0000000..86c06bc --- /dev/null +++ b/Falcon-H1R-7B.Q3_K_L.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c01958f161812aeba48e8c58ee66cc914fd5eb09fb8f069d1462edb3b8a3f08c +size 3916187584 diff --git a/Falcon-H1R-7B.Q3_K_M.gguf b/Falcon-H1R-7B.Q3_K_M.gguf new file mode 100644 index 0000000..764b8ea --- /dev/null +++ b/Falcon-H1R-7B.Q3_K_M.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7fa6fafbdbfb8f603d555949075d6bc70fbfb433b701354268c780589a4bdb51 +size 3687925696 diff --git a/Falcon-H1R-7B.Q3_K_S.gguf b/Falcon-H1R-7B.Q3_K_S.gguf new file mode 100644 index 0000000..44a91e8 --- /dev/null +++ b/Falcon-H1R-7B.Q3_K_S.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:77a72f4ea242b7562a17ecb90581b61d2edab08bfacdcd0abbce9badd1d6f755 +size 3425527744 diff --git a/Falcon-H1R-7B.Q4_K_M.gguf b/Falcon-H1R-7B.Q4_K_M.gguf new file mode 100644 index 0000000..732d035 --- /dev/null +++ b/Falcon-H1R-7B.Q4_K_M.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a7f44537b61815562f9550e5df4867859aff884251d168bffe70a3e7f8c15884 +size 4598344384 diff --git a/Falcon-H1R-7B.Q4_K_S.gguf b/Falcon-H1R-7B.Q4_K_S.gguf new file mode 100644 index 0000000..f5dbb5e --- /dev/null +++ b/Falcon-H1R-7B.Q4_K_S.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8f26d8bf33c303d6cf32fff4819ef6b68eb11c41da9b4a34e14bef4029862a2b +size 4403763904 diff --git a/Falcon-H1R-7B.Q5_K_M.gguf b/Falcon-H1R-7B.Q5_K_M.gguf new file mode 100644 index 0000000..2661c32 --- /dev/null +++ b/Falcon-H1R-7B.Q5_K_M.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e233baa1818112a0360c7895ef60e1b1f96cf903ded2ccd8af900b2063f1c3a2 +size 5390490304 diff --git a/Falcon-H1R-7B.Q5_K_S.gguf b/Falcon-H1R-7B.Q5_K_S.gguf new file mode 100644 index 0000000..9a02920 --- /dev/null +++ b/Falcon-H1R-7B.Q5_K_S.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:213fce94cd55af182b20d83ce9ac18bee6bee0ed92c1ab76eeb2bb01bb227550 +size 5277895360 diff --git a/Falcon-H1R-7B.Q6_K.gguf b/Falcon-H1R-7B.Q6_K.gguf new file mode 100644 index 0000000..22adf66 --- /dev/null +++ b/Falcon-H1R-7B.Q6_K.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6fb8264154c4a11876fdab51e4817d9936c1ed1f913831ca00469fb4f33d62bb +size 6232145344 diff --git a/Falcon-H1R-7B.Q8_0.gguf b/Falcon-H1R-7B.Q8_0.gguf new file mode 100644 index 0000000..b180818 --- /dev/null +++ b/Falcon-H1R-7B.Q8_0.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:94f9f5f2d6c4f4ae850ee18c6727919e79989210a03572cd4ab152e5c44022ca +size 8069003968 diff --git a/Falcon-H1R-7B.f16.gguf b/Falcon-H1R-7B.f16.gguf new file mode 100644 index 0000000..813a466 --- /dev/null +++ b/Falcon-H1R-7B.f16.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0cb11f6f857faaaf757aa63d683d6aaeadfba9bcbc89868c810b85c7fe6e7cc7 +size 15179424448 diff --git a/README.md b/README.md new file mode 100644 index 0000000..d471461 --- /dev/null +++ b/README.md @@ -0,0 +1,85 @@ +--- +license: other +license_name: falcon-llm-license +license_link: https://falconllm.tii.ae/falcon-terms-and-conditions.html +language: +- ar +- cs +- de +- en +- es +- fr +- hi +- it +- ja +- ko +- nl +- pl +- pt +- ro +- ru +- sv +- ur +- zh +tags: +- falcon-h1r +- text-generation-inference +base_model: +- tiiuae/Falcon-H1R-7B +pipeline_tag: text-generation +library_name: transformers +--- + +# **tiiuae_Falcon-H1R-7B-GGUF** + +> Falcon-H1R-7B from TII (Technology Innovation Institute) is a 7-billion-parameter reasoning-specialized causal decoder-only model built on the Falcon-H1-7B-Base foundation, featuring a hybrid Transformer + Mamba2 architecture trained via cold-start supervised fine-tuning with long reasoning traces and scaled RL using GRPO (Generalized Reward Preference Optimization) for exceptional performance in mathematics, programming, instruction following, and general logic. It achieves state-of-the-art results among <8B models across benchmarks like 88.1% on AIME24 (96.7% with test-time scaling), 68.6% on LiveCodeBench v5-v6, 61.3% on GPQA-Diamond, 72.1% on MMLU-Pro, and 53.4% on IFBench—often matching or exceeding 14B-47B competitors like Qwen3-32B, Phi-4-14B, and Nemotron-H-47B while enabling 2x faster inference (e.g., ~1800 tokens/s/GPU at batch=64) and up to 262k context length with low memory footprint. Optimized for multilingual use (English primary, trained on 18 languages including Arabic, Hindi, Chinese) under Falcon-LLM License, it generates structured ... reasoning blocks followed by final answers, deployable via Transformers (temperature=0.6, top_p=0.95, max_new_tokens=65536), vLLM (>=0.11.0, --reasoning-parser deepseek_r1), or SGLang for efficient real-world applications on TP=2 setups. + +## Quick Start with llama-cpp-python + +```py +from llama_cpp import Llama + +llm = Llama.from_pretrained( + repo_id="prithivMLmods/tiiuae_Falcon-H1R-7B-GGUF", + filename="Falcon-H1R-7B.Q4_K_M.gguf", +) +``` + +```py +llm.create_chat_completion( + messages = [ + { + "role": "user", + "content": "What is the capital of France?" + } + ] +) +``` + +## Falcon-H1R-7B [GGUF] + +| File Name | Quant Type | File Size | File Link | +| - | - | - | - | +| Falcon-H1R-7B-bf16.gguf | BF16 | 15.2 GB | [Download](https://huggingface.co/prithivMLmods/tiiuae_Falcon-H1R-7B-GGUF/blob/main/Falcon-H1R-7B-bf16.gguf) | +| Falcon-H1R-7B-f32.gguf | F32 | 30.3 GB | [Download](https://huggingface.co/prithivMLmods/tiiuae_Falcon-H1R-7B-GGUF/blob/main/Falcon-H1R-7B-f32.gguf) | +| Falcon-H1R-7B.IQ4_XS.gguf | IQ4_XS | 4.19 GB | [Download](https://huggingface.co/prithivMLmods/tiiuae_Falcon-H1R-7B-GGUF/blob/main/Falcon-H1R-7B.IQ4_XS.gguf) | +| Falcon-H1R-7B.Q2_K.gguf | Q2_K | 2.89 GB | [Download](https://huggingface.co/prithivMLmods/tiiuae_Falcon-H1R-7B-GGUF/blob/main/Falcon-H1R-7B.Q2_K.gguf) | +| Falcon-H1R-7B.Q3_K_L.gguf | Q3_K_L | 3.92 GB | [Download](https://huggingface.co/prithivMLmods/tiiuae_Falcon-H1R-7B-GGUF/blob/main/Falcon-H1R-7B.Q3_K_L.gguf) | +| Falcon-H1R-7B.Q3_K_M.gguf | Q3_K_M | 3.69 GB | [Download](https://huggingface.co/prithivMLmods/tiiuae_Falcon-H1R-7B-GGUF/blob/main/Falcon-H1R-7B.Q3_K_M.gguf) | +| Falcon-H1R-7B.Q3_K_S.gguf | Q3_K_S | 3.43 GB | [Download](https://huggingface.co/prithivMLmods/tiiuae_Falcon-H1R-7B-GGUF/blob/main/Falcon-H1R-7B.Q3_K_S.gguf) | +| Falcon-H1R-7B.Q4_K_M.gguf | Q4_K_M | 4.6 GB | [Download](https://huggingface.co/prithivMLmods/tiiuae_Falcon-H1R-7B-GGUF/blob/main/Falcon-H1R-7B.Q4_K_M.gguf) | +| Falcon-H1R-7B.Q4_K_S.gguf | Q4_K_S | 4.4 GB | [Download](https://huggingface.co/prithivMLmods/tiiuae_Falcon-H1R-7B-GGUF/blob/main/Falcon-H1R-7B.Q4_K_S.gguf) | +| Falcon-H1R-7B.Q5_K_M.gguf | Q5_K_M | 5.39 GB | [Download](https://huggingface.co/prithivMLmods/tiiuae_Falcon-H1R-7B-GGUF/blob/main/Falcon-H1R-7B.Q5_K_M.gguf) | +| Falcon-H1R-7B.Q5_K_S.gguf | Q5_K_S | 5.28 GB | [Download](https://huggingface.co/prithivMLmods/tiiuae_Falcon-H1R-7B-GGUF/blob/main/Falcon-H1R-7B.Q5_K_S.gguf) | +| Falcon-H1R-7B.Q6_K.gguf | Q6_K | 6.23 GB | [Download](https://huggingface.co/prithivMLmods/tiiuae_Falcon-H1R-7B-GGUF/blob/main/Falcon-H1R-7B.Q6_K.gguf) | +| Falcon-H1R-7B.Q8_0.gguf | Q8_0 | 8.07 GB | [Download](https://huggingface.co/prithivMLmods/tiiuae_Falcon-H1R-7B-GGUF/blob/main/Falcon-H1R-7B.Q8_0.gguf) | +| Falcon-H1R-7B.f16.gguf | F16 | 15.2 GB | [Download](https://huggingface.co/prithivMLmods/tiiuae_Falcon-H1R-7B-GGUF/blob/main/Falcon-H1R-7B.f16.gguf) | + +## Quants Usage + +(sorted by size, not necessarily quality. IQ-quants are often preferable over similar sized non-IQ quants) + +Here is a handy graph by ikawrakow comparing some lower-quality quant +types (lower is better): + +![image.png](https://www.nethype.de/huggingface_embed/quantpplgraph.png) \ No newline at end of file diff --git a/configuration.json b/configuration.json new file mode 100644 index 0000000..bbeeda1 --- /dev/null +++ b/configuration.json @@ -0,0 +1 @@ +{"framework": "pytorch", "task": "text-generation", "allow_remote": true} \ No newline at end of file