Spaces:

lifeofcoding
/

alpaca-lora-movie-review-sentiment

Runtime error

App Files Files Community

lifeofcoding commited on Apr 24, 2023

Commit

955b037

•

1 Parent(s): 3ecbde7

trying new demo

Browse files

Files changed (2) hide show

requirements.txt +8 -2
app.py +113 -23

requirements.txt CHANGED Viewed

@@ -1,2 +1,8 @@
-huggingface-hub
-llama-cpp-python

+datasets
+loralib
+sentencepiece
+git+https://github.com/huggingface/transformers.git
+accelerate
+bitsandbytes
+git+https://github.com/huggingface/peft.git
+gradio

app.py CHANGED Viewed

@@ -4,11 +4,116 @@ import gradio as gr
 from gradio.themes.base import Base
 from gradio.themes.utils import colors, fonts, sizes
-from llama_cpp import Llama
-from huggingface_hub import hf_hub_download
-hf_hub_download(repo_id="lifeofcoding/alpaca-lora-movie-review-sentiment", filename="adapter_model.bin", local_dir=".")
-llm = Llama(model_path="./adapter_model.bin")
 ins = '''Below is an instruction that describes a task. Write a response that appropriately completes the request.
@@ -28,21 +133,6 @@ theme = gr.themes.Monochrome(
-# def generate(instruction):
-#     response = llm(ins.format(instruction))
-#     response = response['choices'][0]['text']
-#     result = ""
-#     for word in response.split(" "):
-#         result += word + " "
-#         yield result
-def generate(instruction):
-    result = ""
-    for x in llm(ins.format(instruction), stop=['### Instruction:', '### End'], stream=True):
-        result += x['choices'][0]['text']
-        yield result
 examples = [
     "Instead of making a peanut butter and jelly sandwich, what else could I combine peanut butter with in a sandwich? Give five ideas",
     "How do I make a campfire?",
@@ -51,7 +141,7 @@ examples = [
 ]
 def process_example(args):
-    for x in generate(args):
         pass
     return x
@@ -137,7 +227,7 @@ with gr.Blocks(theme=seafoam, analytics_enabled=False, css=css) as demo:
-    submit.click(generate, inputs=[instruction], outputs=[output])
-    instruction.submit(generate, inputs=[instruction], outputs=[output])
 demo.queue(concurrency_count=1).launch(debug=True)

 from gradio.themes.base import Base
 from gradio.themes.utils import colors, fonts, sizes
+import torch
+from peft import PeftModel
+import transformers
+assert (
+    "LlamaTokenizer" in transformers._import_structure["models.llama"]
+), "LLaMA is now in HuggingFace's main branch.\nPlease reinstall it: pip uninstall transformers && pip install git+https://github.com/huggingface/transformers.git"
+from transformers import LlamaTokenizer, LlamaForCausalLM, GenerationConfig
+tokenizer = LlamaTokenizer.from_pretrained("decapoda-research/llama-7b-hf")
+BASE_MODEL = "decapoda-research/llama-7b-hf"
+LORA_WEIGHTS = "lifeofcoding/alpaca-lora-movie-review-sentiment"
+if torch.cuda.is_available():
+    device = "cuda"
+else:
+    device = "cpu"
+try:
+    if torch.backends.mps.is_available():
+        device = "mps"
+except:
+    pass
+if device == "cuda":
+    model = LlamaForCausalLM.from_pretrained(
+        BASE_MODEL,
+        load_in_8bit=False,
+        torch_dtype=torch.float16,
+        device_map="auto",
+    )
+    model = PeftModel.from_pretrained(
+        model, LORA_WEIGHTS, torch_dtype=torch.float16, force_download=True
+    )
+elif device == "mps":
+    model = LlamaForCausalLM.from_pretrained(
+        BASE_MODEL,
+        device_map={"": device},
+        torch_dtype=torch.float16,
+    )
+    model = PeftModel.from_pretrained(
+        model,
+        LORA_WEIGHTS,
+        device_map={"": device},
+        torch_dtype=torch.float16,
+    )
+else:
+    model = LlamaForCausalLM.from_pretrained(
+        BASE_MODEL, device_map={"": device}, low_cpu_mem_usage=True
+    )
+    model = PeftModel.from_pretrained(
+        model,
+        LORA_WEIGHTS,
+        device_map={"": device},
+    )
+def generate_prompt(instruction, input=None):
+    if input:
+        return f"""Below is an instruction that describes a task, paired with an input that provides further context. Write a response that appropriately completes the request.
+### Instruction:
+{instruction}
+### Input:
+{input}
+### Response:"""
+    else:
+        return f"""Below is an instruction that describes a task. Write a response that appropriately completes the request.
+### Instruction:
+{instruction}
+### Response:"""
+if device != "cpu":
+    model.half()
+model.eval()
+if torch.__version__ >= "2":
+    model = torch.compile(model)
+def evaluate(
+    instruction,
+    input=None,
+    temperature=0.1,
+    top_p=0.75,
+    top_k=40,
+    num_beams=4,
+    max_new_tokens=128,
+    **kwargs,
+):
+    prompt = generate_prompt(instruction, input)
+    inputs = tokenizer(prompt, return_tensors="pt")
+    input_ids = inputs["input_ids"].to(device)
+    generation_config = GenerationConfig(
+        temperature=temperature,
+        top_p=top_p,
+        top_k=top_k,
+        num_beams=num_beams,
+        **kwargs,
+    )
+    with torch.no_grad():
+        generation_output = model.generate(
+            input_ids=input_ids,
+            generation_config=generation_config,
+            return_dict_in_generate=True,
+            output_scores=True,
+            max_new_tokens=max_new_tokens,
+        )
+    s = generation_output.sequences[0]
+    output = tokenizer.decode(s)
+    return output.split("### Response:")[1].strip()
 ins = '''Below is an instruction that describes a task. Write a response that appropriately completes the request.
 examples = [
     "Instead of making a peanut butter and jelly sandwich, what else could I combine peanut butter with in a sandwich? Give five ideas",
     "How do I make a campfire?",
 ]
 def process_example(args):
+    for x in evaluate(args):
         pass
     return x
+    submit.click(evaluate, inputs=[instruction], outputs=[output])
+    instruction.submit(evaluate, inputs=[instruction], outputs=[output])
 demo.queue(concurrency_count=1).launch(debug=True)