First upload

Browse files

Files changed (10) hide show

README.md +9 -2
added_tokens.json +130 -0
config.json +40 -0
constants_prompt.py +56 -0
example_gptq4bits.py +22 -0
model.safetensors +3 -0
quantize_config.json +11 -0
special_tokens_map.json +24 -0
tokenizer.json +0 -0
tokenizer_config.json +33 -0

README.md CHANGED Viewed

@@ -4,8 +4,15 @@ language:
 ---
 This is a GPTQ 4bits version of Auto-J-13B. We convert it using [this script (by TheBroke)](https://gist.github.com/TheBloke/b47c50a70dd4fe653f64a12928286682#file-quant_autogptq-py).
-It takes about 8GB VRAM to load this model, and we provide an example for using it in `example.py.`
 Note that the behaviours of the quantized model and the original one might be different.
-Please refer to our [github repo](https://github.com/GAIR-NLP/auto-j) for more datails.

 ---
 This is a GPTQ 4bits version of Auto-J-13B. We convert it using [this script (by TheBroke)](https://gist.github.com/TheBloke/b47c50a70dd4fe653f64a12928286682#file-quant_autogptq-py).
+To use the 4bits version of Auto-J, you need to install the following packages:
+```bash
+pip install safetensors
+pip install transformers>=4.32.0 optimum>=1.12.0
+pip install auto-gptq --extra-index-url https://huggingface.github.io/autogptq-index/whl/cu118/  # Use cu117 if on CUDA 11.7
+```
+It takes about 8GB VRAM to load this model, and we provide an example for using it in `example_gptq4bits.py.`
 Note that the behaviours of the quantized model and the original one might be different.
+Please refer to our [github repo](https://github.com/GAIR-NLP/auto-j) for more datails.

added_tokens.json ADDED Viewed

	@@ -0,0 +1,130 @@

+{
+  "<extra_id_32001>": 32001,
+  "<extra_id_32002>": 32002,
+  "<extra_id_32003>": 32003,
+  "<extra_id_32004>": 32004,
+  "<extra_id_32005>": 32005,
+  "<extra_id_32006>": 32006,
+  "<extra_id_32007>": 32007,
+  "<extra_id_32008>": 32008,
+  "<extra_id_32009>": 32009,
+  "<extra_id_32010>": 32010,
+  "<extra_id_32011>": 32011,
+  "<extra_id_32012>": 32012,
+  "<extra_id_32013>": 32013,
+  "<extra_id_32014>": 32014,
+  "<extra_id_32015>": 32015,
+  "<extra_id_32016>": 32016,
+  "<extra_id_32017>": 32017,
+  "<extra_id_32018>": 32018,
+  "<extra_id_32019>": 32019,
+  "<extra_id_32020>": 32020,
+  "<extra_id_32021>": 32021,
+  "<extra_id_32022>": 32022,
+  "<extra_id_32023>": 32023,
+  "<extra_id_32024>": 32024,
+  "<extra_id_32025>": 32025,
+  "<extra_id_32026>": 32026,
+  "<extra_id_32027>": 32027,
+  "<extra_id_32028>": 32028,
+  "<extra_id_32029>": 32029,
+  "<extra_id_32030>": 32030,
+  "<extra_id_32031>": 32031,
+  "<extra_id_32032>": 32032,
+  "<extra_id_32033>": 32033,
+  "<extra_id_32034>": 32034,
+  "<extra_id_32035>": 32035,
+  "<extra_id_32036>": 32036,
+  "<extra_id_32037>": 32037,
+  "<extra_id_32038>": 32038,
+  "<extra_id_32039>": 32039,
+  "<extra_id_32040>": 32040,
+  "<extra_id_32041>": 32041,
+  "<extra_id_32042>": 32042,
+  "<extra_id_32043>": 32043,
+  "<extra_id_32044>": 32044,
+  "<extra_id_32045>": 32045,
+  "<extra_id_32046>": 32046,
+  "<extra_id_32047>": 32047,
+  "<extra_id_32048>": 32048,
+  "<extra_id_32049>": 32049,
+  "<extra_id_32050>": 32050,
+  "<extra_id_32051>": 32051,
+  "<extra_id_32052>": 32052,
+  "<extra_id_32053>": 32053,
+  "<extra_id_32054>": 32054,
+  "<extra_id_32055>": 32055,
+  "<extra_id_32056>": 32056,
+  "<extra_id_32057>": 32057,
+  "<extra_id_32058>": 32058,
+  "<extra_id_32059>": 32059,
+  "<extra_id_32060>": 32060,
+  "<extra_id_32061>": 32061,
+  "<extra_id_32062>": 32062,
+  "<extra_id_32063>": 32063,
+  "<extra_id_32064>": 32064,
+  "<extra_id_32065>": 32065,
+  "<extra_id_32066>": 32066,
+  "<extra_id_32067>": 32067,
+  "<extra_id_32068>": 32068,
+  "<extra_id_32069>": 32069,
+  "<extra_id_32070>": 32070,
+  "<extra_id_32071>": 32071,
+  "<extra_id_32072>": 32072,
+  "<extra_id_32073>": 32073,
+  "<extra_id_32074>": 32074,
+  "<extra_id_32075>": 32075,
+  "<extra_id_32076>": 32076,
+  "<extra_id_32077>": 32077,
+  "<extra_id_32078>": 32078,
+  "<extra_id_32079>": 32079,
+  "<extra_id_32080>": 32080,
+  "<extra_id_32081>": 32081,
+  "<extra_id_32082>": 32082,
+  "<extra_id_32083>": 32083,
+  "<extra_id_32084>": 32084,
+  "<extra_id_32085>": 32085,
+  "<extra_id_32086>": 32086,
+  "<extra_id_32087>": 32087,
+  "<extra_id_32088>": 32088,
+  "<extra_id_32089>": 32089,
+  "<extra_id_32090>": 32090,
+  "<extra_id_32091>": 32091,
+  "<extra_id_32092>": 32092,
+  "<extra_id_32093>": 32093,
+  "<extra_id_32094>": 32094,
+  "<extra_id_32095>": 32095,
+  "<extra_id_32096>": 32096,
+  "<extra_id_32097>": 32097,
+  "<extra_id_32098>": 32098,
+  "<extra_id_32099>": 32099,
+  "<extra_id_32100>": 32100,
+  "<extra_id_32101>": 32101,
+  "<extra_id_32102>": 32102,
+  "<extra_id_32103>": 32103,
+  "<extra_id_32104>": 32104,
+  "<extra_id_32105>": 32105,
+  "<extra_id_32106>": 32106,
+  "<extra_id_32107>": 32107,
+  "<extra_id_32108>": 32108,
+  "<extra_id_32109>": 32109,
+  "<extra_id_32110>": 32110,
+  "<extra_id_32111>": 32111,
+  "<extra_id_32112>": 32112,
+  "<extra_id_32113>": 32113,
+  "<extra_id_32114>": 32114,
+  "<extra_id_32115>": 32115,
+  "<extra_id_32116>": 32116,
+  "<extra_id_32117>": 32117,
+  "<extra_id_32118>": 32118,
+  "<extra_id_32119>": 32119,
+  "<extra_id_32120>": 32120,
+  "<extra_id_32121>": 32121,
+  "<extra_id_32122>": 32122,
+  "<extra_id_32123>": 32123,
+  "<extra_id_32124>": 32124,
+  "<extra_id_32125>": 32125,
+  "<extra_id_32126>": 32126,
+  "<extra_id_32127>": 32127,
+  "<pad>": 32000
+}

config.json ADDED Viewed

	@@ -0,0 +1,40 @@

+{
+  "_name_or_path": "/cpfs01/shared/GAIR/GAIR_hdd/jlli/llama-2-sft/13b/grm/autoj-13b",
+  "architectures": [
+    "LlamaForCausalLM"
+  ],
+  "attention_bias": false,
+  "bos_token_id": 1,
+  "eos_token_id": 2,
+  "hidden_act": "silu",
+  "hidden_size": 5120,
+  "initializer_range": 0.02,
+  "intermediate_size": 13824,
+  "max_position_embeddings": 8192,
+  "model_type": "llama",
+  "num_attention_heads": 40,
+  "num_hidden_layers": 40,
+  "num_key_value_heads": 40,
+  "pad_token_id": 0,
+  "pretraining_tp": 1,
+  "rms_norm_eps": 1e-05,
+  "rope_scaling": null,
+  "rope_theta": 10000.0,
+  "tie_word_embeddings": false,
+  "torch_dtype": "float32",
+  "transformers_version": "4.34.0",
+  "use_cache": true,
+  "vocab_size": 32128,
+  "quantization_config": {
+    "bits": 4,
+    "group_size": 128,
+    "damp_percent": 0.1,
+    "desc_act": true,
+    "static_groups": false,
+    "sym": true,
+    "true_sequential": true,
+    "model_name_or_path": null,
+    "model_file_base_name": null,
+    "quant_method": "gptq"
+  }
+}

constants_prompt.py ADDED Viewed

	@@ -0,0 +1,56 @@

+PROMPT_INPUT_SYSTEM: str = '[INST] <<SYS>>\n{system_message}\n<</SYS>>\n\n{input} [/INST]'
+PROMPT_INPUT_WO_SYSTEM: str = "[INST] {input} [/INST]"
+PROMPT_INPUT_FOR_SCENARIO_CLS: str = "Identify the scenario for the user's query, output 'default' if you are uncertain.\nQuery:\n{input}\nScenario:\n"
+single = """Write critiques for a submitted response on a given user's query, and grade the response:
+[BEGIN DATA]
+***
+[Query]: {prompt}
+***
+[Response]: {response}
+***
+[END DATA]
+Write critiques for this response. After that, you should give a final rating for the response on a scale of 1 to 10 by strictly following this format: "[[rating]]", for example: "Rating: [[5]]"."""
+pairwise_tie = """You are assessing two submitted responses on a given user's query and judging which response is better or they are tied. Here is the data:
+[BEGIN DATA]
+***
+[Query]: {prompt}
+***
+[Response 1]: {response}
+***
+[Response 2]: {response_another}
+***
+[END DATA]
+Here are the instructions to assess and compare the two responses:
+1. Pinpoint the key factors to distinguish these two responses.
+2. Conclude your comparison by providing a final decision on which response is better, or they are tied. Begin your final decision statement with "So, the final decision is Response 1 / Response 2 / Tie". Ensure that your decision aligns coherently with the comprehensive evaluation and comparison you've provided."""
+protocol_mapping = {
+    "pairwise_tie": pairwise_tie,
+    "single": single,
+}
+def llama2_wrapper(usr_msg, sys_msg=None):
+    if sys_msg is None:
+        return PROMPT_INPUT_WO_SYSTEM.format(input=usr_msg)
+    else:
+        return PROMPT_INPUT_SYSTEM.format(input=usr_msg, system_message=sys_msg)
+def build_autoj_input(prompt, resp1, resp2=None, protocol="single"):
+    user_msg = protocol_mapping[protocol].format(prompt=prompt, response=resp1, response_another=resp2)
+    return llama2_wrapper(user_msg, )
+if __name__ == '__main__':
+    t = build_autoj_input("instruction", "resp1", "resp2", "pairwise_tie")
+    print(t)

example_gptq4bits.py ADDED Viewed

	@@ -0,0 +1,22 @@

+from transformers import AutoModelForCausalLM, AutoTokenizer
+from constants_prompt import *
+path = "GAIR/autoj-13b-GPTQ-4bits"
+tokenizer = AutoTokenizer.from_pretrained(path)
+model = AutoModelForCausalLM.from_pretrained(path, device_map="auto")
+query = "<your query>"
+response = "<a response>"
+text = build_autoj_input(query, response)
+# or for pairwise, you can ->
+# response_another = "<another response>"
+# text = build_autoj_input(query, response, response_another, "pairwise_tie")
+inputs = tokenizer(text, return_tensors="pt").to("cuda")
+out = model.generate(**inputs, max_length=1000, temperature=0.0, do_sample=False, top_p=1.0)
+print(tokenizer.decode(out[0], skip_special_tokens=True))
+# note that this output contains the input part, you may need to remove it by yourself

model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:c5763402bd2f563c41445fadb7f825c59f32282ab5d0c84ddc4acc127701fbc9
+size 7920867784

quantize_config.json ADDED Viewed

	@@ -0,0 +1,11 @@

+{
+  "bits": 4,
+  "group_size": 128,
+  "damp_percent": 0.1,
+  "desc_act": true,
+  "static_groups": false,
+  "sym": true,
+  "true_sequential": true,
+  "model_name_or_path": null,
+  "model_file_base_name": null
+}

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,24 @@

+{
+  "bos_token": {
+    "content": "<s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "eos_token": {
+    "content": "</s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": "<pad>",
+  "unk_token": {
+    "content": "<unk>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  }
+}

tokenizer.json ADDED Viewed

The diff for this file is too large to render. See raw diff

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,33 @@

+{
+  "bos_token": {
+    "__type": "AddedToken",
+    "content": "<s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "clean_up_tokenization_spaces": false,
+  "eos_token": {
+    "__type": "AddedToken",
+    "content": "</s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "legacy": false,
+  "model_max_length": 4096,
+  "pad_token": null,
+  "padding_side": "right",
+  "sp_model_kwargs": {},
+  "tokenizer_class": "LlamaTokenizer",
+  "unk_token": {
+    "__type": "AddedToken",
+    "content": "<unk>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  }
+}