add config files

Browse files

Signed-off-by: José Carlos García <hola@josecarlos.me>

Files changed (7) hide show

.gitattributes +1 -0
README.md +144 -5
config.json +36 -0
generation_config.json +6 -0
special_tokens_map.json +41 -0
tokenizer.json +0 -0
tokenizer_config.json +0 -0

.gitattributes CHANGED Viewed

@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text

 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text
+Falcon3-1B-Instruct-1.58bit-q2b0.gguf filter=lfs diff=lfs merge=lfs -text

README.md CHANGED Viewed

@@ -1,5 +1,144 @@
----
-license: other
-license_name: falcon-llm-license
-license_link: https://falconllm.tii.ae/falcon-terms-and-conditions.html
----

+---
+library_name: transformers
+tags:
+- bitnet
+- falcon3
+base_model: tiiuae/Falcon3-1B-Instruct
+license: other
+license_name: falcon-llm-license
+license_link: https://falconllm.tii.ae/falcon-terms-and-conditions.html
+---
+![image/png](https://cdn-uploads.huggingface.co/production/uploads/62441d1d9fdefb55a0b7d12c/c-tosr0FvMlKuKQTojx_6.png)
+#  Table of Contents
+0. [TL;DR](#TL;DR)
+1. [Model Details](#model-details)
+2. [Training Details](#training-details)
+3. [Usage](#usage)
+4. [Evaluation](#evaluation)
+5. [Citation](#citation)
+# TL;DR
+# Model Details
+## Model Description
+- **Developed by:** [https://www.tii.ae](https://www.tii.ae)
+- **Model type:** Causal decoder-only - instruct / chat version
+- **Architecture:** Pure-transformer - 1.58bit version
+- **Language(s) (NLP):** Mainly English
+- **License:** TII Falcon License 2.0
+# Training details
+The model has been trained following the training strategies from the recent [1-bit LLM HF blogpost](https://huggingface.co/blog/1_58_llm_extreme_quantization) and [1-bit LLM paper](https://huggingface.co/papers/2402.17764).
+For more details about the training protocol of this model, please refer to the Falcon-3 technical report, section *Compression*.
+# Usage
+Currently to use this model you can either rely on Hugging Face transformers library or [BitNet](https://github.com/microsoft/BitNet) library. You can also play with the model using the [falcon-1.58bit playground](https://huggingface.co/spaces/tiiuae/falcon3-1.58bit-playground) (only for the 7B instruct version).
+## 🤗 transformers
+```python
+import torch
+from transformers import AutoModelForCausalLM, AutoTokenizer
+model_id = "tiiuae/Falcon3-1B-Instruct-1.58bit"
+model = AutoModelForCausalLM.from_pretrained(
+  model_id,
+  torch_dtype=torch.bfloat16,
+).to("cuda")
+# Perform text generation
+```
+## BitNet
+```
+git clone https://github.com/microsoft/BitNet && cd BitNet
+pip install -r requirements.txt
+python setup_env.py --hf-repo tiiuae/Falcon3-1B-Instruct-1.58bit -q i2_s
+python run_inference.py -m models/Falcon3-1B-1.58bit/ggml-model-i2_s.gguf -p "You are a helpful assistant" -cnv
+```
+# Evaluation
+We report in the following table our internal pipeline benchmarks:
+**Note evaluation results are normalized score from v2 leaderboard tasks - reported results of original models in the blogpost are raw scores**
+<table border="1" style="width: 100%; text-align: center; border-collapse: collapse;">
+    <colgroup>
+        <col style="width: 10%;">
+        <col style="width: 10%;">
+        <col style="background-color: rgba(80, 15, 213, 0.5); width: 7%;">
+    </colgroup>
+    <thead>
+        <tr>
+            <th>Benchmark</th>
+            <th>Llama3-8B-1.58-100B-tokens</th>
+            <th>Falcon3-1B-Instruct-1.58bit</th>
+        </tr>
+    </thead>
+    <tbody>
+        <tr>
+            <td>IFEval</td>
+            <td>17.91</td>
+            <td>44.5</td>
+        </tr>
+        <tr>
+            <td>MUSR</td>
+            <td>4.87</td>
+            <td>2.78</td>
+        </tr>
+        <tr>
+            <td>GPQA</td>
+            <td>1.83</td>
+            <td>0</td>
+        </tr>
+        <tr>
+            <td>BBH</td>
+            <td>5.36</td>
+            <td>2.24</td>
+        </tr>
+        <tr>
+            <td>MMLU-PRO</td>
+            <td>2.78</td>
+            <td>1.93</td>
+        </tr>
+        <tr>
+            <td>MATH</td>
+            <td>0.26</td>
+            <td>0.17</td>
+        </tr>
+        <tr>
+            <td>Average</td>
+            <td>5.5</td>
+            <td>8.6</td>
+        </tr>
+    </tbody>
+</table>
+## Useful links
+- View our [release blogpost](https://huggingface.co/blog/falcon3).
+- Feel free to join [our discord server](https://discord.gg/fwXpMyGc) if you have any questions or to interact with our researchers and developers.
+## Citation
+If the Falcon3 family of models were helpful to your work, feel free to give us a cite.
+```
+@misc{Falcon3,
+    title = {The Falcon 3 Family of Open Models},
+    author = {Falcon-LLM Team},
+    month = {December},
+    year = {2024}
+}
+```

config.json ADDED Viewed

	@@ -0,0 +1,36 @@

+{
+  "_name_or_path": "/home/ec2-user/checkpoints/falcon3-1b-1bit/hf-1bit",
+  "architectures": [
+    "LlamaForCausalLM"
+  ],
+  "attention_bias": false,
+  "attention_dropout": 0.0,
+  "eos_token_id": 11,
+  "head_dim": 256,
+  "hidden_act": "silu",
+  "hidden_size": 2048,
+  "initializer_range": 0.02,
+  "intermediate_size": 8192,
+  "is_bitnet_config": true,
+  "max_position_embeddings": 8192,
+  "mlp_bias": false,
+  "model_type": "llama",
+  "num_attention_heads": 8,
+  "num_hidden_layers": 18,
+  "num_key_value_heads": 4,
+  "pretraining_tp": 1,
+  "quantization_config": {
+    "modules_to_not_convert": [
+      "lm_head"
+    ],
+    "quant_method": "bitnet"
+  },
+  "rms_norm_eps": 1e-06,
+  "rope_scaling": null,
+  "rope_theta": 1000042,
+  "tie_word_embeddings": false,
+  "torch_dtype": "bfloat16",
+  "transformers_version": "4.48.0.dev0",
+  "use_cache": true,
+  "vocab_size": 131072
+}

generation_config.json ADDED Viewed

	@@ -0,0 +1,6 @@

+{
+  "_from_model_config": true,
+  "bos_token_id": 11,
+  "eos_token_id": 11,
+  "transformers_version": "4.48.0.dev0"
+}

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,41 @@

+{
+  "additional_special_tokens": [
+    ">>TITLE<<",
+    ">>ABSTRACT<<",
+    ">>INTRODUCTION<<",
+    ">>SUMMARY<<",
+    ">>COMMENT<<",
+    ">>ANSWER<<",
+    ">>QUESTION<<",
+    ">>DOMAIN<<",
+    ">>EMAIL_ADDRESS<<",
+    ">>IP_ADDRESS<<",
+    "<|startoftext|>",
+    ">>IP_ADDRESS_0<<",
+    ">>IP_ADDRESS_1<<",
+    ">>IP_ADDRESS_2<<",
+    ">>IP_ADDRESS_3<<",
+    ">>IP_ADDRESS_4<<",
+    ">>IP_ADDRESS_5<<",
+    ">>IP_ADDRESS_6<<",
+    ">>IP_ADDRESS_7<<",
+    ">>IP_ADDRESS_8<<",
+    ">>IP_ADDRESS_9<<",
+    ">>PASSWORD<<",
+    ">>KEY<<"
+  ],
+  "eos_token": {
+    "content": "<|endoftext|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": {
+    "content": "<|pad|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  }
+}

tokenizer.json ADDED Viewed

The diff for this file is too large to render. See raw diff

tokenizer_config.json ADDED Viewed

The diff for this file is too large to render. See raw diff