deplot_plus_llm

Runtime error

App Files Files Community

fl399 commited on Apr 5, 2023

Commit

70fa6be

1 Parent(s): 40d28d3

Update app.py

Browse files

Files changed (1) hide show

app.py +97 -58

app.py CHANGED Viewed

@@ -114,19 +114,40 @@ if torch.__version__ >= "2":
 ## FLAN-UL2
-# in dev...
 TOKEN = os.environ.get("API_TOKEN", None)
 API_URL = "https://api-inference.huggingface.co/models/google/flan-ul2"
 headers = {"Authorization": f"Bearer {TOKEN}"}
 def query(payload):
 	response = requests.post(API_URL, headers=headers, json=payload)
 	return response.json()
 def evaluate(
     table,
     question,
     llm="alpaca-lora",
-    num_shot="1-shot",
     input=None,
     temperature=0.1,
     top_p=0.75,
@@ -138,10 +159,7 @@ def evaluate(
     prompt_0shot = _INSTRUCTION + "\n" + _add_markup(table) + "\n" + "Q: " + question + "\n" + "A:"
     prompt = _TEMPLATE + "\n" + _add_markup(table) + "\n" + "Q: " + question + "\n" + "A:"
     if llm == "alpaca-lora":
-        if num_shot == "1-shot":
-            inputs = tokenizer(prompt, return_tensors="pt")
-        else:
-            inputs = tokenizer(prompt_0shot, return_tensors="pt")
         input_ids = inputs["input_ids"].to(device)
         generation_config = GenerationConfig(
             temperature=temperature,
@@ -161,24 +179,15 @@ def evaluate(
         s = generation_output.sequences[0]
         output = tokenizer.decode(s)
     elif llm == "flan-ul2":
-        if num_shot == "1-shot":
-            output = query({
-                "inputs": prompt
-            })[0]["generated_text"]
-        else:
-            output = query({
-                "inputs": prompt_0shot
-            })[0]["generated_text"]
     else:
         RuntimeError(f"No such LLM: {llm}")
     return output
-## deplot models
-model_deplot = Pix2StructForConditionalGeneration.from_pretrained("google/deplot", torch_dtype=torch.bfloat16).to(0)
-processor_deplot = Pix2StructProcessor.from_pretrained("google/deplot")
 def process_document(image, question, llm, num_shot):
     # image = Image.open(image)
     inputs = processor_deplot(images=image, text="Generate the underlying data table for the figure below:", return_tensors="pt").to(0, torch.bfloat16)
@@ -191,45 +200,75 @@ def process_document(image, question, llm, num_shot):
         return [table, res.split("A:")[-1]]
     else:
         return [table, res]
-description = "Demo for DePlot+LLM for QA and summarisation. [DePlot](https://arxiv.org/abs/2212.10505) is an image-to-text model that converts plots and charts into a textual sequence. The sequence then is used to prompt LLM for chain-of-thought reasoning. The current underlying LLMs are [alpaca-lora](https://huggingface.co/spaces/tloen/alpaca-lora) and [flan-ul2](https://huggingface.co/google/flan-ul2). To use it, simply upload your image and type a question or instruction and click 'submit', or click one of the examples to load them. Read more at the links below."
-article = "<p style='text-align: center'><a href='https://arxiv.org/abs/2212.10505' target='_blank'>DePlot: One-shot visual language reasoning by plot-to-table translation</a></p>"
-demo = gr.Interface(
-    fn=process_document,
-    inputs=[
-        "image",
-        "text",
-        gr.Dropdown(
-            ["alpaca-lora", "flan-ul2"], label="LLM", info="Will add more LLMs later!"
-        ),
-        gr.Dropdown(
-            ["0-shot", "1-shot"], label="#shots", info="How many example tables in the prompt?"
-        ),
-    ],
-    outputs=[
-        gr.inputs.Textbox(
-            lines=8,
-            label="Intermediate Table",
-        ),
-        gr.inputs.Textbox(
-            lines=5,
-            label="Output",
-        )
-    ],
-    title="DePlot+LLM (Multimodal chain-of-thought reasoning on plots)",
-    description=description,
-    article=article,
-    enable_queue=True,
-    examples=[["deplot_case_study_m1.png", "What is the sum of numbers of Indonesia and Ireland? Remember to think step by step.", "alpaca-lora", "1-shot"],
-              ["deplot_case_study_m1.png", "Summarise the chart for me please.", "alpaca-lora", "0-shot"],
-              ["deplot_case_study_3.png", "By how much did China's growth rate drop? Think step by step.", "alpaca-lora", "1-shot"],
-              ["deplot_case_study_4.png", "How many papers are submitted in 2020?", "alpaca-lora", "1-shot"],
-              ["deplot_case_study_x2.png", "Summarise the chart for me please.", "alpaca-lora", "0-shot"],
-              ["deplot_case_study_4.png", "How many papers are submitted in 2020?", "flan-ul2", "0-shot"],
-              ["deplot_case_study_4.png", "acceptance rate = # accepted / #submitted . What is the acceptance rate of 2010?", "flan-ul2", "0-shot"],
-              ["deplot_case_study_m1.png", "Summarise the chart for me please.", "flan-ul2", "0-shot"],
              ],
-    cache_examples=True)
 demo.launch(debug=True)

 ## FLAN-UL2
 TOKEN = os.environ.get("API_TOKEN", None)
 API_URL = "https://api-inference.huggingface.co/models/google/flan-ul2"
 headers = {"Authorization": f"Bearer {TOKEN}"}
 def query(payload):
 	response = requests.post(API_URL, headers=headers, json=payload)
 	return response.json()
+## OpenAI models
+def set_openai_api_key(api_key):
+    if api_key and api_key.startswith("sk-") and len(api_key) > 50:
+        openai.api_key = api_key
+def get_response_from_openai(prompt, model="gpt-3.5-turbo", max_output_tokens=128):
+  messages = [{"role": "assistant", "content": prompt}]
+  response = openai.ChatCompletion.create(
+      model=model,
+      messages=messages,
+      temperature=0.7,
+      max_tokens=max_output_tokens,
+      top_p=1,
+      frequency_penalty=0,
+      presence_penalty=0,
+  )
+  ret = response.choices[0].message['content']
+  return ret
+## deplot models
+model_deplot = Pix2StructForConditionalGeneration.from_pretrained("google/deplot", torch_dtype=torch.bfloat16).to(0)
+processor_deplot = Pix2StructProcessor.from_pretrained("google/deplot")
 def evaluate(
     table,
     question,
     llm="alpaca-lora",
     input=None,
     temperature=0.1,
     top_p=0.75,
     prompt_0shot = _INSTRUCTION + "\n" + _add_markup(table) + "\n" + "Q: " + question + "\n" + "A:"
     prompt = _TEMPLATE + "\n" + _add_markup(table) + "\n" + "Q: " + question + "\n" + "A:"
     if llm == "alpaca-lora":
+        inputs = tokenizer(prompt, return_tensors="pt")
         input_ids = inputs["input_ids"].to(device)
         generation_config = GenerationConfig(
             temperature=temperature,
         s = generation_output.sequences[0]
         output = tokenizer.decode(s)
     elif llm == "flan-ul2":
+        output = query({"inputs": prompt_0shot})[0]["generated_text"]
+    elif llm == "gpt-3.5-turbo":
+        output = get_response_from_openai(prompt_0shot)
     else:
         RuntimeError(f"No such LLM: {llm}")
     return output
 def process_document(image, question, llm, num_shot):
     # image = Image.open(image)
     inputs = processor_deplot(images=image, text="Generate the underlying data table for the figure below:", return_tensors="pt").to(0, torch.bfloat16)
         return [table, res.split("A:")[-1]]
     else:
         return [table, res]
+theme = gr.themes.Monochrome(
+    primary_hue="indigo",
+    secondary_hue="blue",
+    neutral_hue="slate",
+    radius_size=gr.themes.sizes.radius_sm,
+    font=[gr.themes.GoogleFont("Open Sans"), "ui-sans-serif", "system-ui", "sans-serif"],
+)
+with gr.Blocks(theme=theme) as demo:
+    with gr.Column():
+      gr.Markdown(
+            """<h1><center>DePlot+LLM: Multimodal chain-of-thought reasoning on plots</center></h1>
+            <p>
+            "This is a demo for DePlot+LLM for QA and summarisation. <a href='https://arxiv.org/abs/2212.10505' target='_blank'>DePlot</a> is an image-to-text model that converts plots and charts into a textual sequence. The sequence then is used to prompt LLM for chain-of-thought reasoning. The current underlying LLMs are <a href='https://huggingface.co/spaces/tloen/alpaca-lora' target='_blank'>alpaca-lora</a> and <a href='https://huggingface.co/google/flan-ul2' target='_blank'>flan-ul2</a>. To use it, simply upload your image and type a question or instruction and click 'submit', or click one of the examples to load them. Read more at the links below."
+            </p>
+            """
+            )
+    # #with gr.Row():
+    # llm = gr.Dropdown(
+    #         ["alpaca-lora", "flan-ul2"], label="LLM", info="We will add more LLMs.")
+    # num_shot = gr.Dropdown(
+    #         ["0-shot", "1-shot"], label="shots", info="How many example tables in the prompt?")
+    # openai_api = gr.Textbox(label="openai api (if using OpenAI models, otherwise leave empty)")
+    with gr.Row():
+      with gr.Column(scale=2):
+        input_image = gr.Image(label="Input Image", type="pil", interactive=True)
+        #input_image.style(height=512, width=512)
+        instruction = gr.Textbox(placeholder="Enter your instruction/question...", label="Question/Instruction")
+        llm = gr.Dropdown(["alpaca-lora", "flan-ul2", "gpt-3.5-turbo"], label="LLM")
+        openai_api_key_textbox = gr.Textbox(placeholder="Paste your OpenAI API key (sk-...) and hit Enter (if using OpenAI models, otherwise leave empty)",
+                                              show_label=False, lines=1, type='password')
+        submit = gr.Button("Submit", variant="primary")
+      with gr.Column(scale=2):
+        with gr.Accordion("Show intermediate table", open=False):
+          output_table = gr.Textbox(lines=8)
+        output_text = gr.Textbox(lines=8,label="Output")
+    gr.Examples(
+    examples=[["deplot_case_study_m1.png", "What is the sum of numbers of Indonesia and Ireland? Remember to think step by step.", "alpaca-lora"],
+              ["deplot_case_study_m1.png", "Summarise the chart for me please.", "alpaca-lora"],
+              ["deplot_case_study_3.png", "By how much did China's growth rate drop? Think step by step.", "alpaca-lora"],
+              ["deplot_case_study_4.png", "How many papers are submitted in 2020?", "alpaca-lora"],
+              ["deplot_case_study_x2.png", "Summarise the chart for me please.", "alpaca-lora"],
+              ["deplot_case_study_4.png", "How many papers are submitted in 2020?", "flan-ul2"],
+              ["deplot_case_study_4.png", "acceptance rate = # accepted / #submitted . What is the acceptance rate of 2010?", "flan-ul2"],
+              ["deplot_case_study_m1.png", "Summarise the chart for me please.", "flan-ul2"],
              ],
+             cache_examples=True,
+             inputs=[input_image, instruction, llm],
+             outputs=[output_table, output_text],
+             fn=process_document
+    )
+    gr.Markdown(
+            """<p style='text-align: center'><a href='https://arxiv.org/abs/2212.10505' target='_blank'>DePlot: One-shot visual language reasoning by plot-to-table translation</a></p>"""
+    )
+    openai_api_key_textbox.change(set_openai_api_key,
+                                      inputs=[openai_api_key_textbox],
+                                      outputs=[])
+    openai_api_key_textbox.submit(set_openai_api_key,
+                                      inputs=[openai_api_key_textbox],
+                                      outputs=[])
+    submit.click(process_document, inputs=[input_image, instruction, llm], outputs=[output_table, output_text])
+    instruction.submit(
+        process_document, inputs=[input_image, instruction, llm], outputs=[output_table, output_text]
+    )
 demo.launch(debug=True)