From 2c7ebaeea5102fea0435f0055241635deabfb307 Mon Sep 17 00:00:00 2001 From: ModelHub XC Date: Sat, 12 Sep 2026 03:52:12 +0800 Subject: [PATCH] =?UTF-8?q?=E5=88=9D=E5=A7=8B=E5=8C=96=E9=A1=B9=E7=9B=AE?= =?UTF-8?q?=EF=BC=8C=E7=94=B1ModelHub=20XC=E7=A4=BE=E5=8C=BA=E6=8F=90?= =?UTF-8?q?=E4=BE=9B=E6=A8=A1=E5=9E=8B?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Model: tiiuae/Falcon-E-3B-Base-prequantized Source: Original Platform --- .gitattributes | 50 ++++++++++ README.md | 218 +++++++++++++++++++++++++++++++++++++++++ chat_template.jinja | 90 +++++++++++++++++ config.json | 32 ++++++ configuration.json | 1 + generation_config.json | 6 ++ model.safetensors | 3 + tokenizer.json | 3 + tokenizer_config.json | 8 ++ 9 files changed, 411 insertions(+) create mode 100644 .gitattributes create mode 100644 README.md create mode 100644 chat_template.jinja create mode 100644 config.json create mode 100644 configuration.json create mode 100644 generation_config.json create mode 100644 model.safetensors create mode 100644 tokenizer.json create mode 100644 tokenizer_config.json diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..98a1fbc --- /dev/null +++ b/.gitattributes @@ -0,0 +1,50 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bin.* filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zstandard filter=lfs diff=lfs merge=lfs -text +*.tfevents* filter=lfs diff=lfs merge=lfs -text +*.db* filter=lfs diff=lfs merge=lfs -text +*.ark* filter=lfs diff=lfs merge=lfs -text +**/*ckpt*data* filter=lfs diff=lfs merge=lfs -text +**/*ckpt*.meta filter=lfs diff=lfs merge=lfs -text +**/*ckpt*.index filter=lfs diff=lfs merge=lfs -text + +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.gguf* filter=lfs diff=lfs merge=lfs -text +*.ggml filter=lfs diff=lfs merge=lfs -text +*.llamafile* filter=lfs diff=lfs merge=lfs -text +*.pt2 filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text + +model.safetensors filter=lfs diff=lfs merge=lfs -text +tokenizer.json filter=lfs diff=lfs merge=lfs -text \ No newline at end of file diff --git a/README.md b/README.md new file mode 100644 index 0000000..bd7d741 --- /dev/null +++ b/README.md @@ -0,0 +1,218 @@ +--- +library_name: transformers +tags: +- bitnet +- falcon-e +- edge +license: other +license_name: falcon-llm-license +license_link: https://falconllm.tii.ae/falcon-terms-and-conditions.html +--- + +![image/png](https://cdn-uploads.huggingface.co/production/uploads/62441d1d9fdefb55a0b7d12c/KVAEDoch-o0HgA0e2L4HL.png) + +# Table of Contents + +0. [TL;DR](#TL;DR) +1. [Model Details](#model-details) +2. [Training Details](#training-details) +3. [Usage](#usage) +4. [Evaluation](#evaluation) +5. [Citation](#citation) + +This is simply the mirror of https://huggingface.co/tiiuae/Falcon-E-3B-Base - branch `prequantized` + +# TL;DR + +# Model Details + +## Model Description + +- **Developed by:** [https://www.tii.ae](https://www.tii.ae) +- **Model type:** Causal decoder-only / Base version +- **Architecture:** Pure-transformer - 1.58bit version +- **Language(s) (NLP):** English +- **License:** Falcon-LLM License + +# Training details + +For more details about the training protocol of this model, please refer to the [Falcon-E technical blogpost](https://falcon-lm.github.io/blog/falcon-edge/). + +# Usage + +Currently to use this model you can either rely on Hugging Face transformers library or [BitNet](https://github.com/microsoft/BitNet) library. There are multiple ways to interact with the model depending on your target usage. For each of the Falcon-E series model, you have three variants: the BitNet model, the prequantized checkpoint for fine-tuning and the `bfloat16` version of the BitNet model. + +### Inference + +#### 🤗 transformers + +In case you want to perform inference on the BitNet checkpoint run: + +```python +import torch +from transformers import AutoModelForCausalLM, AutoTokenizer + +model_id = "tiiuae/Falcon-E-1B-Base" + +model = AutoModelForCausalLM.from_pretrained( + model_id, + torch_dtype=torch.bfloat16, +).to("cuda") + +# Perform text generation +``` + +If you want to rather use the classic `bfloat16` version, you can run: + +```python +import torch +from transformers import AutoModelForCausalLM, AutoTokenizer + +model_id = "tiiuae/Falcon-E-1B-Base" +revision = "bfloat16" + +model = AutoModelForCausalLM.from_pretrained( + model_id, + torch_dtype=torch.bfloat16, + revision=revision, +).to("cuda") + +# Perform text generation +``` + + +#### BitNet + +``` +git clone https://github.com/microsoft/BitNet && cd BitNet +pip install -r requirements.txt +python setup_env.py --hf-repo tiiuae/Falcon-E-1B-Base -q i2_s +python run_inference.py -m models/Falcon-E-1B-Base/ggml-model-i2_s.gguf -p "You are a helpful assistant" -cnv +``` + +#### Apply mlx-lm + +``` +pip install -U mlx-lm +``` + +Then: +``` +mlx_lm.generate --model tiiuae/Falcon-E-3B-Instruct --prompt "Implement bubble sort" --max-tokens 100 --temp 0.1 +``` + + +### Fine-tuning + +For fine-tuning the model, you should load the `prequantized` revision of the model and use the `onebitllms` Python package: + +```diff +import torch + +from transformers import AutoModelForCausalLM, AutoTokenizer +from trl import SFTTrainer ++ from onebitllms import replace_linear_with_bitnet_linear, quantize_to_1bit + +model_id = "tiiuae/Falcon-E-1B-Base" + +tokenizer = AutoTokenizer.from_pretrained(model_id, revision="prequantized") +model = AutoModelForCausalLM.from_pretrained( + model_id, + torch_dtype=torch.bfloat16, ++ revision="prequantized" +) ++ model = replace_linear_with_bitnet_linear(model) + +trainer = SFTTrainer( + model, + ... +) + +trainer.train() + ++ quantize_to_1bit(output_directory) +``` + +# Evaluation + +We report in the following table our internal pipeline benchmarks: + +**Note evaluation results are normalized score from former Hugging Face leaderboard v2 tasks** + +
+ For 1B scale models and below + +| Model | Nb Params | Mem Footprint | IFEVAL | Math-Hard | GPQA | MuSR | BBH | MMLU-Pro | Avg. | +| -------- | ------- | ------- | ------- | ------ | ----- | ----- | ----- | ------ | ---- | +| Qwen-2.5-0.5B | 0.5B | 1GB | 16.27 | 3.93 | 0.0 | 2.08 | 6.95 | 10.06 | 6.55 | +| SmolLM2-360M | 0.36B | 720MB | 21.15 | 1.21 | 0.0 | 7.73 | 5.54 | 1.88 | 6.25 | +| Qwen-2.5-1.5B | 1.5B | 3.1GB | 26.74 | 9.14 | 16.66 | 5.27 | 20.61 | 4.7 | 13.85 | +| Llama-3.2-1B | 1.24B | 2.47GB | 14.78 | 1.21 | 4.37 | 2.56 | 2.26 | 0 | 4.2 | +| SmolLM2-1.7B | 1.7B | 3.4GB | 24.4 | 2.64 | 9.3 | 4.6 | 12.64 | 3.91 | 9.58 | +| Falcon-3-1B-Base | 1.5B | 3GB | 24.28 | 3.32 | 11.34 | 9.71 | 6.76 | 3.91 | 9.89 | +| Hymba-1.5B-Base | 1.5B | 3GB | 22.95 | 1.36 | 7.69 | 5.18 | 10.25 | 0.78 | 8.04 | +| Falcon-E-1B-Base | 1.8B | **635MB** | 32.9 | 10.97 | 2.8 | 3.65 | 12.28 | 17.82 | 13.40 | + +
+ + +
+ For 3B scale models + +| Model | Nb Params | Mem Footprint | IFEVAL | Math-Hard | GPQA | MuSR | BBH | MMLU-Pro | Avg. | +| -------- | ------- | ------- | ------- | ------ | ----- | ----- | ----- | ------ | ---- | +| Falcon-3-3B-Base | 3B | 6.46GB | 15.74 | 11.78 | 21.58 | 6.27 | 18.09 | 6.26 | 15.74 | +| Qwen2.5-3B | 3B | 6.17GB | 26.9 | 14.8 | 24.3 | 11.76 | 24.48 | 6.38 | 18.1 | +| Falcon-E-3B-Base | 3B | **999MB** | 36.67 | 13.45 | 8.67 | 4.14 | 19.83 | 27.16 | 18.32 | + +
+ +Below are the results for instruction fine-tuned models: + +
+ For 1B scale models and below + +| Model | Nb Params | Mem Footprint | IFEVAL | Math-Hard | GPQA | MuSR | BBH | MMLU-Pro | Avg. | +| -------- | ------- | ------- | ------- | ------ | ----- | ----- | ----- | ------ | ---- | +| Qwen-2.5-0.5B-Instruct | 500M | 1GB | 30.71 | 0 | 8.43 | 0.94 | 7.75 | 0 | 6.59 | +| SmolLM2-360M-Instruct | 360M | 720MB | 38.42 | 1.51 | 4.17 | 2.77 | 1.3 | 0.67 | 8.14 | +| Qwen-2.5-1.5B-Instruct | 1.5B | 3.1GB | 44.76 | 22.05 | 19.81 | 3.19 | 19.99 | 0.78 | 18.43 | +| SmolLM2-1.7B | 1.7B | 3.4GB | 53.68 | 5.82 | 10.92 | 4.1 | 11.71 | 0 | 15.02 | +| Falcon-3-1B-Instruct | 1.5B | 3GB | 55.57 | 6.34 | 12.96 | 10.56 | 9.32 | 2.24 | 16.16 | +| Hymba-1.5B-Instruct | 1.5B | 3GB | 60.09 | 2.72 | 4.59 | 1.05 | 11.56 | 5.515 | 14.19 | +| Falcon-E-1B-Instruct | 1.8B | **635MB** | 54.35 | 9.12 | 16.5 | 2.51 | 19.42 | 9.64 | 18.59 | + +
+ + +
+ For 3B scale models + +| Model | Nb Params | Mem Footprint | IFEVAL | Math-Hard | GPQA | MuSR | BBH | MMLU-Pro | Avg. | +| -------- | ------- | ------- | ------- | ------ | ----- | ----- | ----- | ------ | ---- | +| Falcon-3-3B-Instruct | 3B | 6.46GB | 69.77 | 25 | 26.29 | 11.13 | 22.28 | 5.15 | 26.6 | +| Qwen2.5-3B-Instruct | 3B | 6.17GB | 64.75 | 36.78 | 25.8 | 7.57 | 25.05 | 3.02 | 27.16 | +| Falcon-E-3B-Instruct | 3B | **999MB** | 60.97 | 15.3 | 23.59 | 2.12 | 26.45 | 7.45 | 22.64666667 | + +
+ + +## Useful links + +- View [our release blogpost](https://falcon-lm.github.io/blog/falcon-edge/). +- Learn more about [`onebitllms` library](https://github.com/tiiuae/onebitllms). +- Feel free to join [our discord server](https://discord.gg/fwXpMyGc) if you have any questions or to interact with our researchers and developers. + +## Citation + +If the Falcon-E family of models were helpful to your work, feel free to give us a cite. + +``` +@misc{tiionebitllms, + title = {Falcon-E, a series of powerful, universal and fine-tunable 1.58bit language models.}, + author = {Falcon-LLM Team}, + month = {April}, + url = {https://falcon-lm.github.io/blog/falcon-edge}, + year = {2025} +} +``` \ No newline at end of file diff --git a/chat_template.jinja b/chat_template.jinja new file mode 100644 index 0000000..2c25a2e --- /dev/null +++ b/chat_template.jinja @@ -0,0 +1,90 @@ +{# ================= SYSTEM PROMPT ================= #} +{%- if messages and messages[0]['role'] == 'system' %} + {%- set system_msg = messages[0]['content'] %} + {%- set remaining_messages = messages[1:] %} +{%- elif messages and messages[0]['role'] == 'developer' %} + {%- set system_msg = messages[0]['content'] %} + {%- set remaining_messages = messages[1:] %} +{%- else %} + {%- set system_msg = "You are Falcon, a helpful AI assistant created by Technology Innovation Institute (TII)." %} + {%- set remaining_messages = messages %} +{%- endif %} +{%- if tools %} +<|im_start|>system +{{ system_msg }} +# Tools +You may call one or more functions to assist with the user query. + +{%- for tool in tools %} +{{ tool | tojson }} +{%- endfor %} + +<|im_end|> +{%- else %} +<|im_start|>system +{{ system_msg }} +<|im_end|> +{%- endif %} +{# ================= FIND LAST USER QUERY ================= #} +{%- set ns = namespace(multi_step_tool=true, last_query_index=remaining_messages|length - 1) %} +{%- for message in remaining_messages %} + {%- if message.role == "user" %} + {%- set ns.last_query_index = loop.index0 %} + {%- endif %} +{%- endfor %} +{# ================= RENDER MESSAGES ================= #} +{%- for message in remaining_messages %} + {# ---- Normalize content to string ---- #} + {%- set content = message.get('content', '') %} + {%- set reasoning_content = message.get('reasoning_content', none) %} + {%- if content is string %} + {%- set content_str = content %} + {%- elif content is not none %} + {%- set ns_content = namespace(text='') %} + {%- for item in content if item.type == 'text' %} + {%- set ns_content.text = ns_content.text ~ item.text %} + {%- endfor %} + {%- set content_str = ns_content.text %} + {%- endif %} + {# ================= USER ================= #} + {%- if message.role in ['user', 'human'] %} + +<|im_start|>user +{{ content_str }}<|im_end|> + {%- elif message.role in ['assistant', 'gpt', 'function_call'] %} + +<|im_start|>assistant +{%- if reasoning_content is string %} +{{ "\n" ~ reasoning_content ~ "" }} +{%- endif %} + +{{ content_str }} + {%- if message.tool_calls %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + + +{"name": "{{ tool_call.name }}", "arguments": {{ tool_call.arguments if tool_call.arguments is string else (tool_call.arguments | tojson) }}} + + {%- endfor %} + {%- endif %} +<|im_end|> + {%- elif message.role == 'tool' %} + +<|im_start|>user + +{{ content_str }} +<|im_end|> + {%- endif %} +{%- endfor %} +{# ================= GENERATION PROMPT ================= #} +{%- if add_generation_prompt %} + +<|im_start|>assistant +{%- if enable_thinking is defined and enable_thinking is false %} + {{- "\n" }} +{%- endif %} + +{%- endif %} diff --git a/config.json b/config.json new file mode 100644 index 0000000..407a238 --- /dev/null +++ b/config.json @@ -0,0 +1,32 @@ +{ + "architectures": [ + "LlamaForCausalLM" + ], + "attention_bias": false, + "attention_dropout": 0.0, + "bos_token_id": 1, + "dtype": "bfloat16", + "eos_token_id": 11, + "head_dim": 128, + "hidden_act": "silu", + "hidden_size": 2048, + "initializer_range": 0.02, + "intermediate_size": 13312, + "max_position_embeddings": 32768, + "mlp_bias": false, + "model_type": "llama", + "num_attention_heads": 16, + "num_hidden_layers": 32, + "num_key_value_heads": 2, + "pad_token_id": null, + "pretraining_tp": 1, + "rms_norm_eps": 1e-05, + "rope_parameters": { + "rope_theta": 1000000, + "rope_type": "default" + }, + "tie_word_embeddings": false, + "transformers_version": "5.5.0", + "use_cache": true, + "vocab_size": 32768 +} diff --git a/configuration.json b/configuration.json new file mode 100644 index 0000000..bbeeda1 --- /dev/null +++ b/configuration.json @@ -0,0 +1 @@ +{"framework": "pytorch", "task": "text-generation", "allow_remote": true} \ No newline at end of file diff --git a/generation_config.json b/generation_config.json new file mode 100644 index 0000000..7475f0b --- /dev/null +++ b/generation_config.json @@ -0,0 +1,6 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 11, + "transformers_version": "5.5.0" +} diff --git a/model.safetensors b/model.safetensors new file mode 100644 index 0000000..1c1454b --- /dev/null +++ b/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d5bf4792081fb265f2e4c0b17c3b8ea22c8a267a12847e37eaca99d3a5d58ff0 +size 6107206592 diff --git a/tokenizer.json b/tokenizer.json new file mode 100644 index 0000000..cb17d81 --- /dev/null +++ b/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bebbcd0f6c0f52383e4e919d5b24d4585e0203fd010e61999ce339cf4ba2cbef +size 2351345 diff --git a/tokenizer_config.json b/tokenizer_config.json new file mode 100644 index 0000000..9e7cea8 --- /dev/null +++ b/tokenizer_config.json @@ -0,0 +1,8 @@ +{ + "backend": "tokenizers", + "clean_up_tokenization_spaces": true, + "eos_token": "<|end_of_text|>", + "is_local": false, + "model_max_length": 1000000000000000019884624838656, + "tokenizer_class": "TokenizersBackend" +}